Compare commits

..
Author SHA1 Message Date
Ryan Houdek 4a7839b5ac Docs: Update for release FEX-2311.1 2023-11-11 11:59:57 -08:00
Ryan Houdek d8efcb39b8 FEX: Only pass CPU tunables to FEXCore and FEXLoader
This fixes an issue where CPU tunables were ending up in the thunk
generator which means if your CPU doesn't support all the features on
the *Builder* then it would crash with SIGILL. This was happening with
Canonical's runners because they typically only support ARMv8.2 but we
are compiling packages to run on ARMv8.4 devices.

cc: FEX-2311.1
2023-11-11 11:58:25 -08:00
145 changed files with 39957 additions and 33973 deletions

No files matched your search

+1 -6
View File
@@ -38,12 +38,7 @@ check_cxx_source_compiles(
HAS_CLANG_PRESERVE_ALL)
unset(CMAKE_REQUIRED_FLAGS)
if (HAS_CLANG_PRESERVE_ALL)
if (MINGW_BUILD)
message(STATUS "Ignoring broken clang::preserve_all support")
set(HAS_CLANG_PRESERVE_ALL FALSE)
else()
message(STATUS "Has clang::preserve_all")
endif()
message(STATUS "Has clang::preserve_all")
endif ()
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
+1 -1
View File
@@ -652,7 +652,7 @@ def print_ir_allocator_helpers():
# Save NZCV if needed before clobbering NZCV
if op.ImplicitFlagClobber:
output_file.write("\t\tSaveNZCV(IROps::OP_{});".format(op.Name.upper()))
output_file.write("\t\tSaveNZCV();")
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
+3
View File
@@ -90,6 +90,7 @@ set (SRCS
Interface/Core/CPUBackend.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/GdbServer.cpp
Interface/Core/HostFeatures.cpp
Interface/Core/ObjectCache/JobHandling.cpp
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
@@ -100,7 +101,9 @@ set (SRCS
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher/X87F64.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/SignalDelegator.cpp
Interface/Core/X86Tables.cpp
Interface/Core/X86DebugInfo.cpp
Interface/Core/X86HelperGen.cpp
Interface/Core/ArchHelpers/Arm64Emitter.cpp
Interface/Core/Dispatcher/Dispatcher.cpp
@@ -321,6 +321,16 @@ namespace DefaultValues {
Meta->Load();
// Do configuration option fix ups after everything is reloaded
{
// Always fix up the number of threads and create the configuration
// Otherwise the application could receive zero as the number of threads
FEX_CONFIG_OPT(Cores, THREADS);
if (Cores == 0) {
// When the number of emulated CPU cores is zero then auto detect
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, fextl::fmt::format("{}", FEXCore::CPUInfo::CalculateNumberOfCPUs()));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
// Sanitize Core option
FEX_CONFIG_OPT(Core, CORE);
@@ -31,6 +31,15 @@
"Maximum number of instruction to store in a block"
]
},
"Threads": {
"Type": "uint32",
"Default": "0",
"ShortArg": "T",
"Desc": [
"Number of physical hardware threads to tell the process we have.",
"0 will auto detect."
]
},
"CacheObjectCodeCompilation": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
@@ -26,6 +26,12 @@ namespace FEXCore::Context {
return fextl::make_unique<FEXCore::Context::ContextImpl>();
}
bool FEXCore::Context::ContextImpl::InitializeContext() {
// This should be used for generating things that are shared between threads
CPUID.Init(this);
return true;
}
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
CustomExitHandler = std::move(handler);
}
@@ -46,10 +52,22 @@ namespace FEXCore::Context {
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
}
FEXCore::Context::ExitReason FEXCore::Context::ContextImpl::GetExitReason() {
return ParentThread->ExitReason;
}
bool FEXCore::Context::ContextImpl::IsDone() const {
return IsPaused();
}
void FEXCore::Context::ContextImpl::GetCPUState(FEXCore::Core::CPUState *State) const {
memcpy(State, ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
}
void FEXCore::Context::ContextImpl::SetCPUState(const FEXCore::Core::CPUState *State) {
memcpy(ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
}
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
CustomCPUFactory = std::move(Factory);
}
+33 -15
View File
@@ -14,8 +14,8 @@
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/DeferredSignalMutex.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
@@ -37,6 +37,7 @@
namespace FEXCore {
class CodeLoader;
class ThunkHandler;
class GdbServer;
namespace CodeSerialize {
class CodeObjectSerializeService;
@@ -72,6 +73,8 @@ namespace FEXCore::Context {
class ContextImpl final : public FEXCore::Context::Context {
public:
// Context base class implementation.
bool InitializeContext() override;
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
void SetExitHandler(ExitHandler handler) override;
@@ -89,8 +92,15 @@ namespace FEXCore::Context {
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
int GetProgramStatus() const override;
ExitReason GetExitReason() override;
bool IsDone() const override;
void GetCPUState(FEXCore::Core::CPUState *State) const override;
void SetCPUState(const FEXCore::Core::CPUState *State) override;
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
HostFeatures GetHostFeatures() const override;
@@ -98,37 +108,31 @@ namespace FEXCore::Context {
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) override;
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) override;
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param InitialRIP The starting RIP of this thread
* @param StackPointer The starting RSP of this thread
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
* @param NewThreadState The initial thread state to setup for our state
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* Parent thread Creation:
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
* - CTX->RunUntilExit(Thread);
* OS thread Creation:
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread = CreateThread(NewState, PPID);
* - InitializeThread(Thread);
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
* - Thread = CreateThread(CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread = CreateThread(NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
@@ -286,13 +290,17 @@ namespace FEXCore::Context {
void WaitForIdle() override;
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
void StartGdbServer();
void StopGdbServer();
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
return Fn(Frame, record);
}
@@ -303,7 +311,7 @@ namespace FEXCore::Context {
auto Thread = Frame->Thread;
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
@@ -416,6 +424,15 @@ namespace FEXCore::Context {
}
private:
/**
* @brief Does some final thread initialization
*
* @param Thread The internal FEX thread state object
*
* InitCore and CreateThread both call this to finish up thread object initialization
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Initializes the JIT compilers for the thread
*
@@ -433,6 +450,7 @@ namespace FEXCore::Context {
// Entry Cache
std::mutex ExitMutex;
fextl::unique_ptr<GdbServer> DebugServer;
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
@@ -1,6 +1,5 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "FEXCore/Core/X86Enums.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
@@ -29,7 +28,7 @@ namespace FEXCore::CPU {
namespace x64 {
// All but x19 and x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
@@ -37,23 +36,23 @@ namespace x64 {
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29,
// PF/AF must be last.
REG_PF, REG_AF,
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
};
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
// All these callee saved
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
FEXCore::ARMEmitter::Reg::r30,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
}};
// All are caller saved
@@ -176,20 +175,19 @@ namespace x64 {
namespace x32 {
// All but x19 and x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 10> SRA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
// PF/AF must be last.
REG_PF, REG_AF,
};
constexpr std::array<FEXCore::ARMEmitter::Register, 15> RA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
// All these callee saved
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
// Registers only available on 32-bit
// All these are caller saved (except for r19).
@@ -201,10 +199,11 @@ namespace x32 {
FEXCore::ARMEmitter::Reg::r19,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 7> RAPair = {{
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
@@ -369,7 +368,7 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
GeneralFPRegisters = x64::RAFPR;
}
else {
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
StaticRegisters = x32::SRA;
GeneralRegisters = x32::RA;
@@ -400,15 +399,6 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
Segments = 2;
}
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
movn(s, Reg.W(), (~Constant) & 0xFFFF);
if (NOPPad) {
nop(); nop(); nop();
}
return;
}
int RequiredMoveSegments{};
// Count the number of move segments
@@ -591,7 +581,6 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
// Disable FPCR.NEP and FPCR.AH
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
// Also interacts with RPRES to change reciprocal/rsqrt precision from 8-bit mantissa to 12-bit.
//
// Additional interesting AFP bits:
// FIZ(0): Flush Inputs to Zero
@@ -603,23 +592,10 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
}
#endif
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
// is always static and almost certainly clobbered by the subsequent code.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
if (!StaticRegisterAllocation()) {
return;
}
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
GPRSpillMask &= ~PFAFSpillMask;
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i+1];
@@ -635,14 +611,6 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
}
}
// Now handle PF/AF
if (PFAFSpillMask) {
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
}
if (FPRs) {
if (EmitterCTX->HostFeatures.SupportsAVX) {
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
@@ -690,7 +658,7 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
[[maybe_unused]] bool FoundRegister{};
bool FoundRegister{};
for (auto Reg : StaticRegisters) {
if (((1U << Reg.Idx()) & GPRFillMask)) {
TmpReg = Reg;
@@ -720,14 +688,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
}
#endif
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
// is always static and was almost certainly clobbered.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
if (!StaticRegisterAllocation()) {
return;
}
@@ -785,11 +745,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
}
}
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
GPRFillMask &= ~PFAFMask;
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i+1];
@@ -804,14 +759,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
}
}
// Now handle PF/AF
if (PFAFFillMask) {
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
}
}
void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
@@ -53,10 +53,6 @@ constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
@@ -132,21 +128,21 @@ protected:
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
if (SupportsPreserveAllABI) {
SpillForPreserveAllABICall(TmpReg, FPRs);
SpillForPreserveAllABICall(TMP1, true);
}
else {
SpillStaticRegs(TmpReg, FPRs);
PushDynamicRegsAndLR(TmpReg);
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
}
}
void FillForABICall(bool SupportsPreserveAllABI, bool FPRs = true) {
if (SupportsPreserveAllABI) {
FillForPreserveAllABICall(FPRs);
FillForPreserveAllABICall(true);
}
else {
PopDynamicRegsAndLR();
FillStaticRegs(FPRs);
FillStaticRegs();
}
}
@@ -771,16 +771,6 @@ public:
dc32(Op);
}
void axflag() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0101'1111;
dc32(Op);
}
void xaflag() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0011'1111;
dc32(Op);
}
// Conditional compare - register
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0011'1010'010 << 21;
@@ -60,7 +60,7 @@ public:
}
void sha256su1(FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn, FEXCore::ARMEmitter::VRegister rm) {
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
Crypto3RegSHA(Op, 0b110, rd, rn, rm);
Crypto3RegSHA(Op, 0b100, rd, rn, rm);
}
// Cryptographic two-register SHA
+13 -12
View File
@@ -106,10 +106,11 @@ static uint32_t GetCycleCounterFrequency() {
}
void CPUIDEmu::SetupHostHybridFlag() {
PerCPUData.resize(Cores);
size_t CPUs = FEXCore::CPUInfo::CalculateNumberOfCPUs();
PerCPUData.resize(CPUs);
uint64_t MIDR{};
for (size_t i = 0; i < Cores; ++i) {
for (size_t i = 0; i < CPUs; ++i) {
std::error_code ec{};
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
@@ -217,7 +218,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
fextl::vector<const CPUMIDR*> LittleCores;
// Separate CPU cores out to big or little selected
for (size_t i = 0; i < Cores; ++i) {
for (size_t i = 0; i < CPUs; ++i) {
uint32_t MIDR = PerCPUData[i].MIDR;
auto MIDROption = FindDefinedMIDR(MIDR);
if (MIDROption) {
@@ -333,7 +334,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
}
else {
// If we aren't hybrid then just claim everything is big
for (size_t i = 0; i < Cores; ++i) {
for (size_t i = 0; i < CPUs; ++i) {
uint32_t MIDR = PerCPUData[i].MIDR;
auto MIDROption = FindDefinedMIDR(MIDR);
@@ -379,6 +380,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
// Processor Info and Features bits
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
FEXCore::CPUID::FunctionResults Res{};
uint32_t CoreCount = Cores();
// Hypervisor bit is normally set but some applications have issues with it.
uint32_t Hypervisor = HideHypervisorBit() ? 0 : 1;
@@ -387,7 +389,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
Res.ebx = 0 | // Brand index
(8 << 8) | // Cache line size in bytes
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(CoreCount << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(0 << 24); // Local APIC ID
Res.ecx =
@@ -494,7 +496,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
if (Leaf == 0) {
// Report L1D
uint32_t CoreCount = Cores - 1;
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Data | // Cache type
(0b001 << 5) | // Cache level
@@ -518,7 +520,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
}
else if (Leaf == 1) {
// Report L1I
uint32_t CoreCount = Cores - 1;
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Instruction | // Cache type
(0b001 << 5) | // Cache level
@@ -542,7 +544,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
}
else if (Leaf == 2) {
// Report L2
uint32_t CoreCount = Cores - 1;
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Unified | // Cache type
(0b010 << 5) | // Cache level
@@ -566,7 +568,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
}
else if (Leaf == 3) {
// Report L3
uint32_t CoreCount = Cores - 1;
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Unified | // Cache type
(0b011 << 5) | // Cache level
@@ -1068,7 +1070,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) con
(0 << 1) | // IRPerf: Instructions retired count support
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
uint32_t CoreCount = Cores - 1;
uint32_t CoreCount = Cores() - 1;
Res.ecx =
(0 << 16) | // PerfTscSize: Performance timestamp count size
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
@@ -1166,7 +1168,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_001Dh(uint32_t Leaf) con
}
else if (Leaf == 3) {
// Report L3
uint32_t CoreCount = Cores - 1;
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Unified | // Cache type
(0b011 << 5) | // Cache level
@@ -1207,7 +1209,6 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
void CPUIDEmu::Init(FEXCore::Context::ContextImpl *ctx) {
CTX = ctx;
Cores = FEXCore::CPUInfo::CalculateNumberOfCPUs();
// Setup some state tracking
SetupHostHybridFlag();
+1 -1
View File
@@ -113,7 +113,7 @@ public:
private:
FEXCore::Context::ContextImpl *CTX;
bool Hybrid{};
uint32_t Cores{};
FEX_CONFIG_OPT(Cores, THREADS);
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
// XFEATURE_ENABLED_MASK
+118 -42
View File
@@ -9,11 +9,12 @@ $end_info$
*/
#include <cstdint>
#include "FEXCore/Utils/DeferredSignalMutex.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/GdbServer.h"
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/JIT/JITCore.h"
@@ -45,7 +46,6 @@ $end_info$
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/File.h>
#include <FEXCore/Utils/LogManager.h>
#include "FEXCore/Utils/SignalScopeGuards.h"
#include <FEXCore/Utils/Threads.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/fmt.h>
@@ -76,6 +76,64 @@ $end_info$
#include <utility>
#include <xxhash.h>
namespace FEXCore::Core {
struct ThreadLocalData {
FEXCore::Core::InternalThreadState* Thread;
};
constexpr std::array<std::string_view const, 22> FlagNames = {
"CF",
"",
"PF",
"",
"AF",
"",
"ZF",
"SF",
"TF",
"IF",
"DF",
"OF",
"IOPL",
"",
"NT",
"",
"RF",
"VM",
"AC",
"VIF",
"VIP",
"ID",
};
std::string_view const& GetFlagName(unsigned Flag) {
return FlagNames[Flag];
}
constexpr std::array<std::string_view const, 16> RegNames = {
"rax",
"rbx",
"rcx",
"rdx",
"rsi",
"rdi",
"rbp",
"rsp",
"r8",
"r9",
"r10",
"r11",
"r12",
"r13",
"r14",
"r15",
};
std::string_view const& GetGRegName(unsigned Reg) {
return RegNames[Reg];
}
} // namespace FEXCore::Core
namespace FEXCore::Context {
ContextImpl::ContextImpl()
: IRCaptureCache {this} {
@@ -99,8 +157,6 @@ namespace FEXCore::Context {
// Track atomic TSO emulation configuration.
UpdateAtomicTSOEmulationConfig();
CPUID.Init(this);
}
ContextImpl::~ContextImpl() {
@@ -165,7 +221,7 @@ namespace FEXCore::Context {
return Frame->State.rip;
}
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) {
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) {
const auto Frame = Thread->CurrentFrame;
uint32_t EFLAGS{};
@@ -187,23 +243,9 @@ namespace FEXCore::Context {
}
}
// SF/ZF/CF/OF are packed in a 32-bit value in RFLAG_NZCV_LOC.
uint32_t Packed_NZCV{};
if (WasInJIT) {
// If we were in the JIT then NZCV is in the CPU's PSTATE object.
// Packed in to the same bit locations as RFLAG_NZCV_LOC.
Packed_NZCV = PSTATE;
// If we were in the JIT then PF and AF are in registers.
// Move them to the CPUState frame now.
Frame->State.pf_raw = HostGPRs[CPU::REG_PF.Idx()];
Frame->State.af_raw = HostGPRs[CPU::REG_AF.Idx()];
}
else {
// If we were not in the JIT then the NZCV state is stored in the CPUState RFLAG_NZCV_LOC.
// SF/ZF/CF/OF are packed in a 32-bit value in RFLAG_NZCV_LOC.
memcpy(&Packed_NZCV, &Frame->State.flags[X86State::RFLAG_NZCV_LOC], sizeof(Packed_NZCV));
}
memcpy(&Packed_NZCV, &Frame->State.flags[X86State::RFLAG_NZCV_LOC], sizeof(Packed_NZCV));
uint32_t OF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC)) & 1;
uint32_t CF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC)) & 1;
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC)) & 1;
@@ -217,13 +259,13 @@ namespace FEXCore::Context {
// PF calculation is deferred, calculate it now.
// Popcount the 8-bit flag and then extract the lower bit.
uint32_t PFByte = Frame->State.pf_raw & 0xff;
uint32_t PFByte = Frame->State.flags[X86State::RFLAG_PF_RAW_LOC];
uint32_t PF = std::popcount(PFByte ^ 1) & 1;
EFLAGS |= PF << X86State::RFLAG_PF_RAW_LOC;
// AF calculation is deferred, calculate it now.
// XOR with PF byte and extract bit 4.
uint32_t AF = ((Frame->State.af_raw ^ PFByte) & (1 << 4)) ? 1 : 0;
uint32_t AF = ((Frame->State.flags[X86State::RFLAG_AF_RAW_LOC] ^ PFByte) & (1 << 4)) ? 1 : 0;
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
return EFLAGS;
@@ -243,11 +285,11 @@ namespace FEXCore::Context {
// AF stored in bit 4 in our internal representation. It is also
// XORed with byte 4 of the PF byte, but we write that as zero here so
// we don't need any special handling for that.
Frame->State.af_raw = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
break;
case X86State::RFLAG_PF_RAW_LOC:
// PF is inverted in our internal representation.
Frame->State.pf_raw = (EFLAGS & (1U << i)) ? 0 : 1;
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 0 : 1;
break;
default:
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 1 : 0;
@@ -317,6 +359,13 @@ namespace FEXCore::Context {
// Give this configuration to the SignalDelegator.
SignalDelegation->SetConfig(SignalConfig);
if (Config.GdbServer) {
StartGdbServer();
}
else {
StopGdbServer();
}
#ifndef _WIN32
ThunkHandler = FEXCore::ThunkHandler::Create();
#else
@@ -324,21 +373,36 @@ namespace FEXCore::Context {
Config.NeedsPendingInterruptFaultCheck = true;
#endif
if (Config.GdbServer) {
// If gdbserver is enabled then this needs to be enabled.
Config.NeedsPendingInterruptFaultCheck = true;
// FEX needs to start paused when gdb is enabled.
StartPaused = true;
}
using namespace FEXCore::Core;
FEXCore::Core::InternalThreadState *Thread = CreateThread(InitialRIP, StackPointer, nullptr, 0);
FEXCore::Core::InternalThreadState *Thread = CreateThread(nullptr, 0);
// We are the parent thread
ParentThread = Thread;
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
Thread->CurrentFrame->State.rip = InitialRIP;
InitializeThreadData(Thread);
return Thread;
}
void ContextImpl::StartGdbServer() {
#ifndef _WIN32
if (!DebugServer) {
DebugServer = fextl::make_unique<GdbServer>(this, SignalDelegation, SyscallHandler);
StartPaused = true;
}
#endif
}
void ContextImpl::StopGdbServer() {
#ifndef _WIN32
DebugServer.reset();
#endif
}
void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
}
@@ -511,6 +575,14 @@ namespace FEXCore::Context {
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
}
int ContextImpl::GetProgramStatus() const {
return ParentThread->StatusCode;
}
void ContextImpl::InitializeThreadData(FEXCore::Core::InternalThreadState *Thread) {
Thread->CPUBackend->Initialize();
}
struct ExecutionThreadHandler {
ContextImpl *This;
FEXCore::Core::InternalThreadState *Thread;
@@ -609,22 +681,20 @@ namespace FEXCore::Context {
Thread->PassManager->Finalize();
}
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{};
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
Thread->CurrentFrame->State.rip = InitialRIP;
// Copy over the new thread state to the new object
if (NewThreadState) {
memcpy(&Thread->CurrentFrame->State, NewThreadState, sizeof(FEXCore::Core::CPUState));
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
}
Thread->CurrentFrame->Thread = Thread;
// Set up the thread manager state
Thread->ThreadManager.parent_tid = ParentTID;
Thread->CurrentFrame->Thread = Thread;
InitializeCompiler(Thread);
InitializeThreadData(Thread);
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
@@ -1043,7 +1113,7 @@ namespace FEXCore::Context {
auto Thread = Frame->Thread;
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
ScopedDeferredSignalWithForkableSharedLock lk(CodeInvalidationMutex, Thread);
// Is the code in the cache?
// The backends only check L1 and L2, not L3
@@ -1226,7 +1296,7 @@ namespace FEXCore::Context {
// Potential deferred since Thread might not be valid.
// Thread object isn't valid very early in frontend's initialization.
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
auto lk = GuardSignalDeferringSectionWithFallback(CodeInvalidationMutex, Thread);
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
InvalidateGuestCodeRangeInternal(this, Start, Length);
}
@@ -1235,7 +1305,7 @@ namespace FEXCore::Context {
// Potential deferred since Thread might not be valid.
// Thread object isn't valid very early in frontend's initialization.
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
auto lk = GuardSignalDeferringSectionWithFallback(CodeInvalidationMutex, Thread);
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
InvalidateGuestCodeRangeInternal(this, Start, Length);
CallAfter(Start, Length);
@@ -1263,7 +1333,7 @@ namespace FEXCore::Context {
}
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
}
@@ -1311,11 +1381,17 @@ namespace FEXCore::Context {
IR::AOTIRCacheEntry *ContextImpl::LoadAOTIRCacheEntry(const fextl::string &filename) {
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
if (DebugServer) {
DebugServer->AlertLibrariesChanged();
}
return rv;
}
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry) {
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
if (DebugServer) {
DebugServer->AlertLibrariesChanged();
}
}
void ContextImpl::AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
@@ -85,9 +85,7 @@ public:
#endif
uint16_t GetSRAGPRCount() const {
// PF/AF are the final two SRA registers.
// Only return the SRA for GPRs.
return StaticRegisters.size() - 2;
return StaticRegisters.size();
}
uint16_t GetSRAFPRCount() const {
@@ -95,7 +93,7 @@ public:
}
void GetSRAGPRMapping(uint8_t Mapping[16]) const {
for (size_t i = 0; i < StaticRegisters.size() - 2; ++i) {
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
Mapping[i] = StaticRegisters[i].Idx();
}
}
@@ -12,7 +12,6 @@ $end_info$
#include <memory>
#include <optional>
#include <Common/FEXServerClient.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CodeLoader.h>
#include <FEXCore/Core/Context.h>
@@ -33,7 +32,6 @@ $end_info$
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/Filesystem.h>
#include <atomic>
#include <cstring>
@@ -44,72 +42,18 @@ $end_info$
#endif
#include <errno.h>
#include <fcntl.h>
#include <poll.h>
#include <fmt/format.h>
#include <signal.h>
#include <stddef.h>
#include <string_view>
#include <sys/stat.h>
#include <sys/un.h>
#include <sys/utsname.h>
#include <unistd.h>
#include <utility>
#include "LinuxSyscalls/GdbServer.h"
#include "GdbServer.h"
namespace FEX
namespace FEXCore
{
constexpr std::array<std::string_view const, 22> FlagNames = {
"CF",
"",
"PF",
"",
"AF",
"",
"ZF",
"SF",
"TF",
"IF",
"DF",
"OF",
"IOPL",
"",
"NT",
"",
"RF",
"VM",
"AC",
"VIF",
"VIP",
"ID",
};
static std::string_view const& GetFlagName(unsigned Flag) {
return FlagNames[Flag];
}
static std::string_view const GetGRegName(unsigned Reg) {
switch (Reg) {
case FEXCore::X86State::REG_RAX: return "rax";
case FEXCore::X86State::REG_RBX: return "rbx";
case FEXCore::X86State::REG_RCX: return "rcx";
case FEXCore::X86State::REG_RDX: return "rdx";
case FEXCore::X86State::REG_RSP: return "rsp";
case FEXCore::X86State::REG_RBP: return "rbp";
case FEXCore::X86State::REG_RSI: return "rsi";
case FEXCore::X86State::REG_RDI: return "rdi";
case FEXCore::X86State::REG_R8: return "r8";
case FEXCore::X86State::REG_R9: return "r9";
case FEXCore::X86State::REG_R10: return "r10";
case FEXCore::X86State::REG_R11: return "r11";
case FEXCore::X86State::REG_R12: return "r12";
case FEXCore::X86State::REG_R13: return "r13";
case FEXCore::X86State::REG_R14: return "r14";
case FEXCore::X86State::REG_R15: return "r15";
default: FEX_UNREACHABLE;
}
}
#ifndef _WIN32
void GdbServer::Break(int signal) {
std::lock_guard lk(sendMutex);
@@ -126,12 +70,7 @@ void GdbServer::WaitForThreadWakeup() {
ThreadBreakEvent.Wait();
}
GdbServer::~GdbServer() {
CoreShuttingDown = true;
close(ListenSocket);
}
GdbServer::GdbServer(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler)
GdbServer::GdbServer(FEXCore::Context::Context *ctx, SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler)
: CTX(ctx)
, SyscallHandler {SyscallHandler} {
// Pass all signals by default
@@ -149,7 +88,7 @@ GdbServer::GdbServer(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *
// This is a total hack as there is currently no way to resume once hitting a segfault
// But it's semi-useful for debugging.
for (uint32_t Signal = 0; Signal <= FEX::HLE::SignalDelegator::MAX_SIGNALS; ++Signal) {
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
SignalDelegation->RegisterHostSignalHandler(Signal, [this] (FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
if (PassSignals[Signal]) {
// Pass signal to the guest
@@ -201,10 +140,6 @@ static fextl::string encodeHex(const unsigned char *data, size_t length) {
return ss.str();
}
static fextl::string encodeHex(std::string_view str) {
return encodeHex(reinterpret_cast<const unsigned char*>(str.data()), str.size());
}
static fextl::string getThreadName(uint32_t ThreadID) {
const auto ThreadFile = fextl::fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
fextl::string ThreadName;
@@ -319,15 +254,15 @@ struct X80Float {
};
struct FEX_PACKED GDBContextDefinition {
uint64_t gregs[FEXCore::Core::CPUState::NUM_GPRS];
uint64_t gregs[Core::CPUState::NUM_GPRS];
uint64_t rip;
uint32_t eflags;
uint32_t cs, ss, ds, es, fs, gs;
X80Float mm[FEXCore::Core::CPUState::NUM_MMS];
X80Float mm[Core::CPUState::NUM_MMS];
uint32_t fctrl;
uint32_t fstat;
uint32_t dummies[6];
uint64_t xmm[FEXCore::Core::CPUState::NUM_XMMS][4];
uint64_t xmm[Core::CPUState::NUM_XMMS][4];
uint32_t mxcsr;
};
@@ -358,9 +293,9 @@ fextl::string GdbServer::readRegs() {
memcpy(&GDB.gregs[0], &state.gregs[0], sizeof(GDB.gregs));
memcpy(&GDB.rip, &state.rip, sizeof(GDB.rip));
GDB.eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread, false, nullptr, 0);
GDB.eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread);
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
memcpy(&GDB.mm[i], &state.mm[i], sizeof(GDB.mm));
}
@@ -414,7 +349,7 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
return {encodeHex((unsigned char *)(&state.rip), sizeof(uint64_t)), HandledPacketType::TYPE_ACK};
}
else if (addr == offsetof(GDBContextDefinition, eflags)) {
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread, false, nullptr, 0);
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread);
return {encodeHex((unsigned char *)(&eflags), sizeof(uint32_t)), HandledPacketType::TYPE_ACK};
}
@@ -448,9 +383,9 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
}
else if (addr >= offsetof(GDBContextDefinition, xmm[0][0]) &&
addr < offsetof(GDBContextDefinition, xmm[16][0])) {
const auto XmmIndex = (addr - offsetof(GDBContextDefinition, xmm[0][0])) / FEXCore::Core::CPUState::XMM_AVX_REG_SIZE;
const auto XmmIndex = (addr - offsetof(GDBContextDefinition, xmm[0][0])) / Core::CPUState::XMM_AVX_REG_SIZE;
const auto *Data = (unsigned char *)&state.xmm.avx.data[XmmIndex][0];
return {encodeHex(Data, FEXCore::Core::CPUState::XMM_AVX_REG_SIZE), HandledPacketType::TYPE_ACK};
return {encodeHex(Data, Core::CPUState::XMM_AVX_REG_SIZE), HandledPacketType::TYPE_ACK};
}
else if (addr == offsetof(GDBContextDefinition, mxcsr)) {
uint32_t Empty{};
@@ -474,7 +409,7 @@ fextl::string buildTargetXML() {
xml << "<flags id='fex_eflags' size='4'>\n";
// flags register
for(int i = 0; i < 22; i++) {
auto name = GetFlagName(i);
auto name = FEXCore::Core::GetFlagName(i);
if (name.empty()) {
continue;
}
@@ -492,8 +427,8 @@ fextl::string buildTargetXML() {
// We want to just memcpy our x86 state to gdb, so we tell it the ordering.
// GPRs
for (uint32_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; i++) {
reg(GetGRegName(i), "int64", 64);
for (uint32_t i = 0; i < Core::CPUState::NUM_GPRS; i++) {
reg(FEXCore::Core::GetGRegName(i), "int64", 64);
}
reg("rip", "code_ptr", 64);
@@ -549,7 +484,7 @@ fextl::string buildTargetXML() {
)";
// SSE regs
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
reg(fextl::fmt::format("xmm{}", i), "vec128", 128);
}
@@ -575,7 +510,7 @@ fextl::string buildTargetXML() {
<field name="uint128" type="uint128"/>
</union>
)";
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
reg(fmt::format("ymm{}h", i), "vec128", 128);
}
xml << "</feature>\n";
@@ -880,6 +815,7 @@ GdbServer::HandledPacketType GdbServer::handleMemory(const fextl::string &packet
}
}
GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet) {
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
const auto MatchStr = [](const fextl::string &Str, const char *str) -> bool { return Str.rfind(str, 0) == 0; };
@@ -931,17 +867,12 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
SupportedFeatures += "QNonStop+;";
SupportedFeatures += "qXfer:osdata:read+;";
SupportedFeatures += "QStartNoAckMode+;";
// TODO: Support breakpoints
// SupportedFeatures += "swbreak+;";
// SupportedFeatures += "hwbreak+;";
// SupportedFeatures += "BreakpointCommands+;";
// TODO: If we want to support conditional breakpoints then we need to support single stepping.
// SupportedFeatures += "ConditionalBreakpoints+;";
// Causes GDB to crash?
// SupportedFeatures += "QStartNoAckMode+;";
for (auto &Feature : Features) {
if (MatchStr(Feature, "swbreak+")) {
SupportedFeatures += "swbreak+;";
}
@@ -1041,68 +972,13 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
// We now have a semi-colon deliminated list of signals to pass to the guest process
for (fextl::string tmp; std::getline(ss, tmp, ';'); ) {
uint32_t Signal = std::stoi(tmp.c_str(), nullptr, 16);
if (Signal < FEX::HLE::SignalDelegator::MAX_SIGNALS) {
if (Signal < SignalDelegator::MAX_SIGNALS) {
PassSignals[Signal] = true;
}
}
return {"OK", HandledPacketType::TYPE_ACK};
}
// lldb specific queries
if (match("qHostInfo")) {
// Returns Key:Value pairs separated by ;
// eg:
// triple:7838365f36342d70632d6c696e75782d676e75;
// ptrsize:8;
// distribution_id:7562756e7475;
// watchpoint_exceptions_received:after;
// endian:little;
// os_version:6.3.3;
// os_build:362e332e332d3036303330332d67656e65726963;
// os_kernel:2332303233303531373133333620534d5020505245454d50545f44594e414d494320576564204d61792031372031333a34353a3139205554432032303233;
// hostname:7279616e682d545235303030;
fextl::string HostFeatures{};
// 64-bit always returned for the host environment.
// qProcessInfo will return i386 or not.
HostFeatures += fextl::fmt::format("triple:{};", encodeHex("x86_64-pc-linux-gnu"));
HostFeatures += "ptrsize:8;";
// Always little-endian.
HostFeatures += "endian:little;";
struct utsname buf{};
if (uname(&buf) != -1) {
uint32_t Major{};
uint32_t Minor{};
uint32_t Patch{};
// Parse kernel version in the form of `<Major>.<Minor>.<Patch>[Optional Data]`
const auto End = buf.release + sizeof(buf.release);
auto Results = std::from_chars(buf.release, End, Major, 10);
Results = std::from_chars(Results.ptr + 1, End, Minor, 10);
Results = std::from_chars(Results.ptr + 1, End, Patch, 10);
HostFeatures += fextl::fmt::format("os_version:{}.{}.{};", Major, Minor, Patch);
// os_build returns the release untouched.
HostFeatures += fextl::fmt::format("os_build:{};", encodeHex(buf.release));
HostFeatures += fextl::fmt::format("os_kernel:{};", encodeHex(buf.version));
HostFeatures += fextl::fmt::format("hostname:{};", encodeHex(buf.nodename));
}
// TODO: distribution_id should be fetched with `lsb_release -i`
// TODO: watchpoint_exceptions_received is unsupported
return {std::move(HostFeatures), HandledPacketType::TYPE_ACK};
}
if (match("qGetWorkingDir")) {
char Tmp[PATH_MAX];
if (getcwd(Tmp, PATH_MAX)) {
return {encodeHex(Tmp), HandledPacketType::TYPE_ACK};
}
return {"E00", HandledPacketType::TYPE_ACK};
}
return {"", HandledPacketType::TYPE_UNKNOWN};
}
@@ -1345,45 +1221,11 @@ void GdbServer::SendPacketPair(const HandledPacketType& response) {
}
}
GdbServer::WaitForConnectionResult GdbServer::WaitForConnection() {
while (!CoreShuttingDown.load()) {
struct pollfd PollFD {
.fd = ListenSocket,
.events = POLLIN | POLLPRI | POLLRDHUP,
.revents = 0,
};
int Result = ppoll(&PollFD, 1, nullptr, nullptr);
if (Result > 0) {
if (PollFD.revents & POLLIN) {
CommsStream = OpenSocket();
return WaitForConnectionResult::CONNECTION;
}
else if (PollFD.revents & (POLLHUP | POLLERR | POLLNVAL)) {
// Listen socket error or shutting down
LogMan::Msg::EFmt("[GdbServer] gdbserver shutting down: {}");
return WaitForConnectionResult::ERROR;
}
}
else if (Result == -1) {
LogMan::Msg::EFmt("[GdbServer] poll failure: {}", errno);
}
}
LogMan::Msg::EFmt("[GdbServer] Shutting Down");
return WaitForConnectionResult::ERROR;
}
void GdbServer::GdbServerLoop() {
OpenListenSocket();
if (ListenSocket == -1) {
// Couldn't open socket, just exit.
return;
}
while (!CoreShuttingDown.load()) {
if (WaitForConnection() == WaitForConnectionResult::ERROR) {
break;
}
CommsStream = OpenSocket();
HandledPacketType response{};
@@ -1433,11 +1275,9 @@ void GdbServer::GdbServerLoop() {
}
close(ListenSocket);
unlink(GdbUnixSocketPath.c_str());
}
static void* ThreadHandler(void *Arg) {
FEXCore::Threads::SetThreadName("FEX:gdbserver");
auto This = reinterpret_cast<FEX::GdbServer*>(Arg);
FEXCore::GdbServer *This = reinterpret_cast<FEXCore::GdbServer*>(Arg);
This->GdbServerLoop();
return nullptr;
}
@@ -1449,48 +1289,38 @@ void GdbServer::StartThread() {
}
void GdbServer::OpenListenSocket() {
const auto GdbUnixPath = fextl::fmt::format("{}/FEX_gdbserver/", FEXServerClient::GetTempFolder());
if (FHU::Filesystem::CreateDirectory(GdbUnixPath) == FHU::Filesystem::CreateDirectoryResult::ERROR) {
LogMan::Msg::EFmt("[GdbServer] Couldn't create gdbserver folder {}", GdbUnixPath);
return;
// getaddrinfo allocates memory that can't be removed.
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
struct addrinfo hints, *res;
memset(&hints, 0, sizeof(hints));
hints.ai_family = AF_UNSPEC;
hints.ai_socktype = SOCK_STREAM;
hints.ai_flags = AI_PASSIVE;
if(getaddrinfo(NULL, "8086", &hints, &res) < 0) {
perror("getaddrinfo");
}
GdbUnixSocketPath = fextl::fmt::format("{}{}-gdb", GdbUnixPath, ::getpid());
int on = 1;
ListenSocket = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
if (ListenSocket == -1) {
LogMan::Msg::EFmt("[GdbServer] Couldn't open AF_UNIX socket {} {}", errno, strerror(errno));
return;
ListenSocket = socket(res->ai_family, res->ai_socktype, res->ai_protocol);
if (ListenSocket < 0) {
perror("socket");
}
struct sockaddr_un addr{};
addr.sun_family = AF_UNIX;
strncpy(addr.sun_path, GdbUnixSocketPath.data(), sizeof(addr.sun_path));
size_t SizeOfAddr = offsetof(sockaddr_un, sun_path) + GdbUnixSocketPath.size();
// Bind the socket to the path
int Result{};
for (int attempt = 0; attempt < 2; ++attempt) {
Result = bind(ListenSocket, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr);
if (Result == 0) {
break;
}
// This can happen periodically with execve. unlink the path and try again.
// The PID is reused but FEX likely started a gdbserver thread for the PID before execve.
unlink(GdbUnixSocketPath.c_str());
if(setsockopt(ListenSocket, SOL_SOCKET, SO_REUSEADDR, (char*)&on, sizeof(on)) < 0) {
perror("setsockopt");
close(ListenSocket);
}
if (Result != 0) {
LogMan::Msg::EFmt("[GdbServer] Couldn't bind AF_UNIX socket '{}': {} {}\n", addr.sun_path, errno, strerror(errno));
if (bind(ListenSocket, res->ai_addr, res->ai_addrlen) < 0) {
perror("bind");
close(ListenSocket);
ListenSocket = -1;
return;
}
listen(ListenSocket, 1);
LogMan::Msg::IFmt("[GdbServer] Waiting for connection on {}", GdbUnixSocketPath);
LogMan::Msg::IFmt("[GdbServer] gdb-multiarch -ex \"target extended-remote {}\"", GdbUnixSocketPath);
freeaddrinfo(res);
}
fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
@@ -1498,6 +1328,7 @@ fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
struct sockaddr_storage their_addr{};
socklen_t addr_size{};
LogMan::Msg::IFmt("GdbServer, waiting for connection on localhost:8086");
int new_fd = accept(ListenSocket, (struct sockaddr *)&their_addr, &addr_size);
return fextl::make_unique<FEXCore::Utils::NetStream>(new_fd);
@@ -19,14 +19,11 @@ $end_info$
#include <mutex>
#include <stdint.h>
#include "LinuxSyscalls/SignalDelegator.h"
namespace FEX {
namespace FEXCore {
class GdbServer {
public:
GdbServer(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler);
~GdbServer();
GdbServer(FEXCore::Context::Context *ctx, SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler);
// Public for threading
void GdbServerLoop();
@@ -39,11 +36,6 @@ private:
void Break(int signal);
void OpenListenSocket();
enum class WaitForConnectionResult {
CONNECTION,
ERROR,
};
WaitForConnectionResult WaitForConnection();
fextl::unique_ptr<std::iostream> OpenSocket();
void StartThread();
fextl::string ReadPacket(std::iostream &stream);
@@ -98,10 +90,9 @@ private:
fextl::string LibraryMapString{};
// Used to keep track of which signals to pass to the guest
std::array<bool, FEX::HLE::SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
uint32_t CurrentDebuggingThread{};
int ListenSocket{};
fextl::string GdbUnixSocketPath{};
FEX_CONFIG_OPT(Filename, APP_FILENAME);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
};
@@ -180,13 +180,11 @@ static void OverrideFeatures(HostFeatures *Features) {
if (EnableCrypto) {
Features->SupportsAES = true;
Features->SupportsCRC = true;
Features->SupportsSHA = true;
Features->SupportsPMULL_128Bit = true;
}
else if (DisableCrypto) {
Features->SupportsAES = false;
Features->SupportsCRC = false;
Features->SupportsSHA = false;
Features->SupportsPMULL_128Bit = false;
}
if (EnableRPRES) {
@@ -213,8 +211,6 @@ HostFeatures::HostFeatures() {
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) &&
Features.Has(vixl::CPUFeatures::Feature::kSHA2);
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
@@ -242,14 +238,13 @@ HostFeatures::HostFeatures() {
#endif
// TODO: AVX2 is currently unsupported. Disable until the remaining features are implemented.
SupportsAVX2 = false;
SupportsSHA = true;
SupportsBMI1 = true;
SupportsBMI2 = true;
SupportsCLWB = true;
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
SupportsAFP = false;
// RPRES has a dependency on AFP. Disable it until AFP is enabled.
SupportsRPRES = false;
if (!SupportsAtomics) {
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
@@ -288,8 +283,6 @@ HostFeatures::HostFeatures() {
#ifdef VIXL_SIMULATOR
// simulator doesn't support dc(ZVA)
SupportsCLZERO = false;
// Simulator doesn't support SHA
SupportsSHA = false;
#else
// Check if we can support cacheline clears
uint32_t DCZID = GetDCZID();
+157 -143
View File
@@ -87,7 +87,7 @@ DEF_OP(Add) {
DEF_OP(AddNZCV) {
auto Op = IROp->C<IR::IROp_AddNZCV>();
const auto OpSize = IROp->Size;
const IR::OpSize OpSize = Op->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
@@ -98,63 +98,78 @@ DEF_OP(AddNZCV) {
} else {
cmn(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
// TODO: Optimize this out
mrs(GetReg(Node), ARMEmitter::SystemRegister::NZCV);
}
DEF_OP(AdcNZCV) {
auto Op = IROp->C<IR::IROp_AdcNZCV>();
const auto OpSize = IROp->Size;
const IR::OpSize OpSize = Op->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto Dst = GetReg(Node);
// TODO: Optimize this out
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->NZCV.ID()));
adcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
// TODO: Optimize this out
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
}
DEF_OP(SbbNZCV) {
auto Op = IROp->C<IR::IROp_SbbNZCV>();
const auto OpSize = IROp->Size;
const IR::OpSize OpSize = Op->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto Dst = GetReg(Node);
// Carry-in needs to be inverted for subtractions due to carry versus borrow
// distinction between x86 and arm.
// See below remarks on cfinv
eor(ARMEmitter::Size::i32Bit, TMP1, GetReg(Op->NZCV.ID()), 1u << 29);
// TODO: Optimize this out
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
sbcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
// TODO: Optimize this out
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
// The carry flag produced by arm64 sbcs is inverted compared to the x86 carry
// flag. Invert it now.
//
// TODO: Once we optimize out the mrs, this will become a cfinv operation, but
// that's only available with Feat_FlagM. For now the portable way is to flip
// bit 29 (carry) manually.
eor(ARMEmitter::Size::i32Bit, Dst, Dst, 1u << 29);
}
DEF_OP(TestNZ) {
auto Op = IROp->C<IR::IROp_TestNZ>();
const uint8_t OpSize = IROp->Size;
const uint8_t OpSize = Op->Size;
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
uint64_t Const;
auto Src1 = GetReg(Op->Src1.ID());
const auto Dst = GetReg(Node);
auto Src = GetReg(Op->Src1.ID());
// Shift the sign bit into place, clearing out the garbage in upper bits.
// Adding zero does an effective test, setting NZ according to the result and
// zeroing CV.
// setf+rmif would avoid the scratch register, but higher latency on M1.
if (OpSize < 4) {
// Cheaper to and+cmn than to lsl+lsl+tst, so do the and ourselves if
// needed.
if (Op->Src1 != Op->Src2) {
if (IsInlineConstant(Op->Src2, &Const)) {
and_(EmitSize, TMP1, Src1, Const);
} else {
auto Src2 = GetReg(Op->Src2.ID());
and_(EmitSize, TMP1, Src1, Src2);
}
Src1 = TMP1;
}
unsigned Shift = 32 - (OpSize * 8);
cmn(EmitSize, ARMEmitter::Reg::zr, Src1, ARMEmitter::ShiftType::LSL, Shift);
} else {
if (IsInlineConstant(Op->Src2, &Const)) {
tst(EmitSize, Src1, Const);
} else {
const auto Src2 = GetReg(Op->Src2.ID());
tst(EmitSize, Src1, Src2);
}
lsl(EmitSize, Dst, Src, 32 - (OpSize * 8));
Src = Dst;
}
tst(EmitSize, Src, Src);
// TODO: Optimize this out
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
}
DEF_OP(Sub) {
@@ -184,7 +199,7 @@ DEF_OP(SubShift) {
DEF_OP(SubNZCV) {
auto Op = IROp->C<IR::IROp_SubNZCV>();
const auto OpSize = IROp->Size;
const IR::OpSize OpSize = Op->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
@@ -198,70 +213,20 @@ DEF_OP(SubNZCV) {
} else {
cmp(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
}
DEF_OP(CarryInvert) {
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
cfinv();
}
const auto Dst = GetReg(Node);
DEF_OP(RmifNZCV) {
auto Op = IROp->C<IR::IROp_RmifNZCV>();
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
// TODO: Optimize this out
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
rmif(GetReg(Op->Src.ID()).X(), Op->Rotate, Op->Mask);
}
DEF_OP(AXFlag) {
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM2, "Unsupported flagm2 op");
axflag();
}
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
return ARMEmitter::Condition::CC_NV;
}
}
DEF_OP(CondAddNZCV) {
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
ARMEmitter::StatusFlags Flags = (ARMEmitter::StatusFlags)Op->FalseNZCV;
uint64_t Const = 0;
auto Src1 = IsInlineConstant(Op->Src1, &Const) ? ARMEmitter::Reg::zr :
GetReg(Op->Src1.ID());
LOGMAN_THROW_A_FMT(Const == 0, "Unsupported inline constant");
if (IsInlineConstant(Op->Src2, &Const)) {
ccmn(EmitSize, Src1, Const, Flags, MapSelectCC(Op->Cond));
} else {
ccmn(EmitSize, Src1, GetReg(Op->Src2.ID()), Flags, MapSelectCC(Op->Cond));
if (Op->InvertCarry) {
// The carry flag produced by arm64 subs is inverted compared to the x86 carry
// flag. Invert it now.
//
// TODO: Once we optimize out the mrs, this will become a cfinv operation, but
// that's only available with Feat_FlagM. For now the portable way is to flip
// bit 29 (carry) manually.
eor(ARMEmitter::Size::i32Bit, Dst, Dst, 1u << 29);
}
}
@@ -272,10 +237,27 @@ DEF_OP(Neg) {
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
if (Op->Cond == FEXCore::IR::COND_AL)
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
else
cneg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()), MapSelectCC(Op->Cond));
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
}
DEF_OP(Abs) {
auto Op = IROp->C<IR::IROp_Abs>();
const uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto Dst = GetReg(Node);
auto Src = GetReg(Op->Src.ID());
if (CTX->HostFeatures.SupportsCSSC) {
// On CSSC supporting processors, this turns in to one instruction and doesn't modify flags.
abs(EmitSize, Dst, Src);
}
else {
cmp(EmitSize, Src, 0);
cneg(EmitSize, Dst, Src, ARMEmitter::Condition::CC_MI);
}
}
DEF_OP(Mul) {
@@ -577,16 +559,6 @@ DEF_OP(Xor) {
}
}
DEF_OP(XorShift) {
auto Op = IROp->C<IR::IROp_XorShift>();
const uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
eor(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
DEF_OP(Lshl) {
auto Op = IROp->C<IR::IROp_Lshl>();
const uint8_t OpSize = IROp->Size;
@@ -1335,6 +1307,36 @@ DEF_OP(Sbfe) {
sbfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
}
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_ANDZ:
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
case FEXCore::IR::COND_ANDNZ:
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
case FEXCore::IR::COND_MI:
case FEXCore::IR::COND_PL:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
return ARMEmitter::Condition::CC_NV;
}
}
DEF_OP(Select) {
auto Op = IROp->C<IR::IROp_Select>();
const uint8_t OpSize = IROp->Size;
@@ -1343,15 +1345,28 @@ DEF_OP(Select) {
uint64_t Const;
auto cc = MapSelectCC(Op->Cond);
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
Op->Cond == FEXCore::IR::COND_ANDNZ;
LOGMAN_THROW_A_FMT(!tests || IsGPR(Op->Cmp1.ID()), "Only GPRs can be tested");
if (IsGPR(Op->Cmp1.ID())) {
const auto Src1 = GetReg(Op->Cmp1.ID());
if (IsInlineConstant(Op->Cmp2, &Const))
cmp(CompareEmitSize, Src1, Const);
else {
const auto Src2 = GetReg(Op->Cmp2.ID());
cmp(CompareEmitSize, Src1, Src2);
if (tests) {
if (IsInlineConstant(Op->Cmp2, &Const))
tst(CompareEmitSize, Src1, Const);
else {
const auto Src2 = GetReg(Op->Cmp2.ID());
tst(CompareEmitSize, Src1, Src2);
}
} else {
if (IsInlineConstant(Op->Cmp2, &Const))
cmp(CompareEmitSize, Src1, Const);
else {
const auto Src2 = GetReg(Op->Cmp2.ID());
cmp(CompareEmitSize, Src1, Src2);
}
}
}
else if (IsGPRPair(Op->Cmp1.ID())) {
@@ -1390,38 +1405,6 @@ DEF_OP(Select) {
}
}
DEF_OP(NZCVSelect) {
auto Op = IROp->C<IR::IROp_NZCVSelect>();
const uint8_t OpSize = IROp->Size;
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
auto cc = MapSelectCC(Op->Cond);
uint64_t const_true, const_false;
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
uint64_t all_ones = OpSize == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
ARMEmitter::Register Dst = GetReg(Node);
if (is_const_true) {
if (is_const_false != true || !(const_true == 1 || const_true == all_ones) || const_false != 0) {
LOGMAN_MSG_A_FMT("NZCVSelect: Unsupported constant");
}
if (const_true == all_ones)
csetm(EmitSize, Dst, cc);
else
cset(EmitSize, Dst, cc);
} else if (is_const_false) {
LOGMAN_THROW_A_FMT(const_false == 0, "NZCVSelect: unsupported constant");
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), ARMEmitter::Reg::zr, cc);
} else {
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), GetReg(Op->FalseVal.ID()), cc);
}
}
DEF_OP(VExtractToGPR) {
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
const auto OpSize = IROp->Size;
@@ -1536,10 +1519,41 @@ DEF_OP(FCmp) {
auto Op = IROp->C<IR::IROp_FCmp>();
const auto EmitSubSize = Op->ElementSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit;
ARMEmitter::Register Dst = GetReg(Node);
ARMEmitter::VRegister Scalar1 = GetVReg(Op->Scalar1.ID());
ARMEmitter::VRegister Scalar2 = GetVReg(Op->Scalar2.ID());
fcmp(EmitSubSize, Scalar1, Scalar2);
bool set = false;
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ)) {
LOGMAN_THROW_AA_FMT(IR::FCMP_FLAG_EQ == 0, "IR::FCMP_FLAG_EQ must equal 0");
// EQ or unordered
cset(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Condition::CC_EQ); // Z = 1
csinc(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::zr, ARMEmitter::Condition::CC_VC); // IF !V ? Z : 1
set = true;
}
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
// LT or unordered
cset(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Condition::CC_LT);
if (!set) {
lsl(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_LT);
set = true;
} else {
bfi(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_LT, 1);
}
}
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED)) {
cset(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Condition::CC_VS);
if (!set) {
lsl(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_UNORDERED);
set = true;
} else {
bfi(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_UNORDERED, 1);
}
}
}
#undef DEF_OP
@@ -97,7 +97,9 @@ DEF_OP(Jump) {
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_ANDZ:
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
case FEXCore::IR::COND_ANDNZ:
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
@@ -115,8 +117,8 @@ static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
case FEXCore::IR::COND_MI:
case FEXCore::IR::COND_PL:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
return ARMEmitter::Condition::CC_NV;
@@ -128,27 +130,42 @@ DEF_OP(CondJump) {
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
if (Op->FromNZCV) {
b(MapBranchCC(Op->Cond), TrueTargetLabel);
} else {
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
Op->Cond == FEXCore::IR::COND_ANDNZ;
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
const auto SubSize = ARMEmitter::ToVectorSizePair(Op->CompareSize == 4 ? ARMEmitter::SubRegSize::i32Bit : ARMEmitter::SubRegSize::i64Bit);
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ ||
Op->Cond.Val == FEXCore::IR::COND_NEQ,
"CondJump: Expected simple condition");
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
} else {
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
} else if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_NEQ) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
} else {
if (IsGPR(Op->Cmp1.ID())) {
if (tests) {
if (isConst) {
tst(Size, GetReg(Op->Cmp1.ID()), Const);
} else {
tst(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
}
} else {
if (isConst) {
cmp(Size, GetReg(Op->Cmp1.ID()), Const);
} else {
cmp(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
}
}
} else if (IsFPR(Op->Cmp1.ID())) {
fcmp(SubSize.Scalar, GetVReg(Op->Cmp1.ID()), GetVReg(Op->Cmp2.ID()));
} else {
LOGMAN_MSG_A_FMT("CondJump: Expected GPR or FPR");
}
// TODO: Wire up tbz/tbnz
b(MapBranchCC(Op->Cond), TrueTargetLabel);
}
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
@@ -179,32 +179,6 @@ DEF_OP(CRC32) {
}
}
DEF_OP(VSha1H) {
auto Op = IROp->C<IR::IROp_VSha1H>();
const auto Dst = GetVReg(Node);
const auto Src = GetVReg(Op->Src.ID());
sha1h(Dst.S(), Src.S());
}
DEF_OP(VSha256U0) {
auto Op = IROp->C<IR::IROp_VSha256U0>();
const auto Dst = GetVReg(Node);
const auto Src1 = GetVReg(Op->Src1.ID());
const auto Src2 = GetVReg(Op->Src2.ID());
if (Dst == Src1) {
sha256su0(Dst, Src2);
}
else {
mov(VTMP1.Q(), Src1.Q());
sha256su0(VTMP1, Src2);
mov(Dst.Q(), Src1.Q());
}
}
DEF_OP(PCLMUL) {
const auto Op = IROp->C<IR::IROp_PCLMUL>();
const auto OpSize = IROp->Size;
@@ -5,7 +5,6 @@ tags: backend|arm64
$end_info$
*/
#include "FEXCore/Core/X86Enums.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
@@ -296,11 +295,7 @@ DEF_OP(LoadRegisterSRA) {
const auto OpSize = IROp->Size;
if (Op->Class == IR::GPRClass) {
const auto regId =
Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ? (StaticRegisters.size() - 2) :
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ? (StaticRegisters.size() - 1) :
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
const auto regOffs = Op->Offset & 7;
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
@@ -478,14 +473,10 @@ DEF_OP(StoreRegisterSRA) {
const auto OpSize = IROp->Size;
if (Op->Class == IR::GPRClass) {
const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
const auto regOffs = Op->Offset & 7;
const auto regId =
Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ? (StaticRegisters.size() - 2) :
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ? (StaticRegisters.size() - 1) :
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
const auto reg = StaticRegisters[regId];
const auto Src = GetReg(Op->Value.ID());
@@ -1040,37 +1031,23 @@ DEF_OP(FillRegister) {
}
}
DEF_OP(LoadNZCV) {
auto Dst = GetReg(Node);
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
}
DEF_OP(StoreNZCV) {
auto Op = IROp->C<IR::IROp_StoreNZCV>();
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value.ID()));
}
DEF_OP(LoadFlag) {
auto Op = IROp->C<IR::IROp_LoadFlag>();
auto Dst = GetReg(Node);
LOGMAN_THROW_A_FMT(Op->Flag != X86State::RFLAG_PF_RAW_LOC &&
Op->Flag != X86State::RFLAG_AF_RAW_LOC,
"PF/AF must be accessed as registers");
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
if (Op->Flag == 24 /* NZCV */)
ldr(Dst.W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
else
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
}
DEF_OP(StoreFlag) {
auto Op = IROp->C<IR::IROp_StoreFlag>();
LOGMAN_THROW_A_FMT(Op->Flag != X86State::RFLAG_PF_RAW_LOC &&
Op->Flag != X86State::RFLAG_AF_RAW_LOC,
"PF/AF must be accessed as registers");
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
if (Op->Flag == 24 /* NZCV */)
str(GetReg(Op->Value.ID()).W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
else
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
}
FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize,
@@ -1863,15 +1840,9 @@ DEF_OP(MemSet) {
const auto MemReg = GetReg(Op->Addr.ID());
const auto Value = GetReg(Op->Value.ID());
const auto Length = GetReg(Op->Length.ID());
const auto Direction = GetReg(Op->Direction.ID());
const auto Dst = GetReg(Node);
uint64_t DirectionConstant;
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
if (!DirectionIsInline) {
DirectionReg = GetReg(Op->Direction.ID());
}
// If Direction == 0 then:
// MemReg is incremented (by size)
// else:
@@ -1891,10 +1862,8 @@ DEF_OP(MemSet) {
add(TMP2, Prefix.X(), MemReg.X());
}
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
}
// Backward or forwards implementation depends on flag
cbnz(ARMEmitter::Size::i64Bit, Direction, &BackwardImpl);
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
switch (OpSize) {
@@ -1948,7 +1917,8 @@ DEF_OP(MemSet) {
}
};
auto EmitMemset = [&](int32_t Direction) {
// Emit forward direction memset then backward direction memset.
for (int32_t Direction : { 1, -1 }) {
const int32_t OpSize = Size;
const int32_t SizeDirection = Size * Direction;
@@ -2008,26 +1978,15 @@ DEF_OP(MemSet) {
break;
}
}
};
if (DirectionIsInline) {
// If the direction constant is set then the direction is negative.
EmitMemset(DirectionConstant ? -1 : 1);
}
else {
// Emit forward direction memset then backward direction memset.
for (int32_t Direction : { 1, -1 }) {
EmitMemset(Direction);
if (Direction == 1) {
b(&Done);
Bind(&BackwardImpl);
}
if (Direction == 1) {
b(&Done);
Bind(&BackwardImpl);
}
Bind(&Done);
// Destination already set to the final pointer.
}
Bind(&Done);
// Destination already set to the final pointer.
}
DEF_OP(MemCpy) {
@@ -2042,12 +2001,7 @@ DEF_OP(MemCpy) {
const auto MemRegSrc = GetReg(Op->AddrSrc.ID());
const auto Length = GetReg(Op->Length.ID());
uint64_t DirectionConstant;
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
if (!DirectionIsInline) {
DirectionReg = GetReg(Op->Direction.ID());
}
const auto Direction = GetReg(Op->Direction.ID());
auto Dst = GetRegPair(Node);
// If Direction == 0 then:
@@ -2084,10 +2038,8 @@ DEF_OP(MemCpy) {
// TMP3 = Src
// TMP4 = load+store temp value
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
}
// Backward or forwards implementation depends on flag
cbnz(ARMEmitter::Size::i64Bit, Direction, &BackwardImpl);
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
switch (OpSize) {
@@ -2209,7 +2161,8 @@ DEF_OP(MemCpy) {
}
};
auto EmitMemcpy = [&](int32_t Direction) {
// Emit forward direction memset then backward direction memset.
for (int32_t Direction : { 1, -1 }) {
const int32_t OpSize = Size;
const int32_t SizeDirection = Size * Direction;
@@ -2282,24 +2235,15 @@ DEF_OP(MemCpy) {
break;
}
}
};
if (DirectionIsInline) {
// If the direction constant is set then the direction is negative.
EmitMemcpy(DirectionConstant ? -1 : 1);
}
else {
// Emit forward direction memset then backward direction memset.
for (int32_t Direction : { 1, -1 }) {
EmitMemcpy(Direction);
if (Direction == 1) {
b(&Done);
Bind(&BackwardImpl);
}
if (Direction == 1) {
b(&Done);
Bind(&BackwardImpl);
}
Bind(&Done);
// Destination already set to the final pointer.
}
Bind(&Done);
// Destination already set to the final pointer.
}
DEF_OP(ParanoidLoadMemTSO) {
@@ -1799,9 +1799,6 @@ DEF_OP(VFMin) {
}
} else {
if (IsScalar) {
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
switch (ElementSize) {
case 2: {
fcmp(Vector1.H(), Vector2.H());
@@ -1821,9 +1818,6 @@ DEF_OP(VFMin) {
default:
break;
}
// Restore NZCV
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
} else {
if (Dst == Vector1) {
// Destination is already Vector1, need to insert Vector2 on false.
@@ -1884,9 +1878,6 @@ DEF_OP(VFMax) {
}
} else {
if (IsScalar) {
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
switch (ElementSize) {
case 2: {
fcmp(Vector1.H(), Vector2.H());
@@ -1906,9 +1897,6 @@ DEF_OP(VFMax) {
default:
break;
}
// Restore NZCV
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
} else {
if (Dst == Vector1) {
// Destination is already Vector1, need to insert Vector2 on true.
@@ -2666,9 +2654,6 @@ DEF_OP(VCMPEQ) {
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i128Bit);
if (HostSupportsSVE256 && Is256Bit) {
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
const auto Mask = PRED_TMP_32B.Zeroing();
const auto ComparePred = ARMEmitter::PReg::p0;
@@ -2679,9 +2664,6 @@ DEF_OP(VCMPEQ) {
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
// Restore NZCV
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
} else {
if (IsScalar) {
cmeq(SubRegSize.Scalar, Dst, Vector1, Vector2);
@@ -2713,9 +2695,6 @@ DEF_OP(VCMPEQZ) {
const auto Mask = PRED_TMP_32B.Zeroing();
const auto ComparePred = ARMEmitter::PReg::p0;
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
// Ensure no junk is in the temp (important for ensuring
// non-equal entries remain as zero).
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
@@ -2726,9 +2705,6 @@ DEF_OP(VCMPEQZ) {
cmpeq(SubRegSize.Vector, ComparePred, Mask, Vector.Z(), 0);
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector.Z());
mov(Dst.Z(), VTMP1.Z());
// Restore NZCV
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
} else {
if (IsScalar) {
cmeq(SubRegSize.Scalar, Dst, Vector);
@@ -2761,9 +2737,6 @@ DEF_OP(VCMPGT) {
const auto Mask = PRED_TMP_32B.Zeroing();
const auto ComparePred = ARMEmitter::PReg::p0;
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
// General idea is to compare for greater-than, bitwise NOT
// the valid values, then ORR the NOTed values with the original
// values to form entries that are all 1s.
@@ -2771,9 +2744,6 @@ DEF_OP(VCMPGT) {
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
// Restore NZCV
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
} else {
if (IsScalar) {
cmgt(SubRegSize.Scalar, Dst, Vector1, Vector2);
@@ -2805,9 +2775,6 @@ DEF_OP(VCMPGTZ) {
const auto Mask = PRED_TMP_32B.Zeroing();
const auto ComparePred = ARMEmitter::PReg::p0;
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
// Ensure no junk is in the temp (important for ensuring
// non greater-than values remain as zero).
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
@@ -2815,9 +2782,6 @@ DEF_OP(VCMPGTZ) {
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector.Z());
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector.Z());
mov(Dst.Z(), VTMP1.Z());
// Restore NZCV
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
} else {
if (IsScalar) {
cmgt(SubRegSize.Scalar, Dst, Vector);
@@ -2849,9 +2813,6 @@ DEF_OP(VCMPLTZ) {
const auto Mask = PRED_TMP_32B.Zeroing();
const auto ComparePred = ARMEmitter::PReg::p0;
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
// Ensure no junk is in the temp (important for ensuring
// non less-than values remain as zero).
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
@@ -2859,9 +2820,6 @@ DEF_OP(VCMPLTZ) {
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector.Z());
orr(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), VTMP1.Z(), Vector.Z());
mov(Dst.Z(), VTMP1.Z());
// Restore NZCV
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
} else {
if (IsScalar) {
cmlt(SubRegSize.Scalar, Dst, Vector);
@@ -3946,58 +3904,6 @@ DEF_OP(VUShrI) {
}
}
DEF_OP(VUShraI) {
const auto Op = IROp->C<IR::IROp_VUShraI>();
const auto OpSize = IROp->Size;
const auto BitShift = Op->BitShift;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetVReg(Node);
const auto DestVector = GetVReg(Op->DestVector.ID());
const auto Vector = GetVReg(Op->Vector.ID());
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize =
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
if (HostSupportsSVE256 && Is256Bit) {
if (Dst == DestVector) {
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
}
else {
if (Dst != Vector) {
mov(Dst.Z(), DestVector.Z());
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
}
else {
mov(VTMP1.Z(), DestVector.Z());
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
mov(Dst.Z(), VTMP1.Z());
}
}
} else {
if (Dst == DestVector) {
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
}
else {
if (Dst != Vector) {
mov(Dst.Q(), DestVector.Q());
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
}
else {
mov(VTMP1.Q(), DestVector.Q());
usra(SubRegSize, VTMP1.Q(), Vector.Q(), BitShift);
mov(Dst.Q(), VTMP1.Q());
}
}
}
}
DEF_OP(VSShrI) {
const auto Op = IROp->C<IR::IROp_VSShrI>();
const auto OpSize = IROp->Size;
File diff suppressed because it is too large. Load diff
+78 -274
View File
@@ -38,13 +38,6 @@ enum class MemoryAccessType {
STREAM,
};
enum class BTAction {
BTNone,
BTClear,
BTSet,
BTComplement,
};
struct LoadSourceOptions {
// Alignment of the load in bytes. -1 signifies unaligned
int8_t Align = -1;
@@ -96,6 +89,7 @@ public:
TYPE_RORI,
TYPE_ROL,
TYPE_ROLI,
TYPE_FCMP,
TYPE_BEXTR,
TYPE_BLSI,
TYPE_BLSMSK,
@@ -104,6 +98,7 @@ public:
TYPE_BZHI,
TYPE_TZCNT,
TYPE_LZCNT,
TYPE_BITSELECT,
TYPE_RDRAND,
};
@@ -127,10 +122,11 @@ public:
}
void StartNewBlock() {
flagsOp = SelectionFlag::Nothing;
// If we loaded flags but didn't change them, invalidate the cached copy and move on.
// Changes get stored out by CalculateDeferredFlags.
CachedNZCV = nullptr;
PossiblySetNZCVBits = ~0U;
// New block needs to reset segment telemetry.
SegmentsNeedReadCheck = ~0U;
@@ -159,14 +155,6 @@ public:
CalculateDeferredFlags();
return _CondJump(ssa0, ssa1, ssa2, cond);
}
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
CalculateDeferredFlags();
// The jump will ignore the sources, so it doesn't matter what we put here.
// Put an inline constant so RA+codegen will ignore altogether.
auto Placeholder = _InlineConstant(0);
return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true);
}
bool FinishOp(uint64_t NextRIP, bool LastOp) {
// If we are switching to a new block and this current block has yet to set a RIP
@@ -336,10 +324,14 @@ public:
void RCLOp1Bit(OpcodeArgs);
void RCLOp(OpcodeArgs);
void RCLSmallerOp(OpcodeArgs);
template<uint32_t SrcIndex, enum BTAction Action>
template<uint32_t SrcIndex>
void BTOp(OpcodeArgs);
template<uint32_t SrcIndex>
void BTROp(OpcodeArgs);
template<uint32_t SrcIndex>
void BTSOp(OpcodeArgs);
template<uint32_t SrcIndex>
void BTCOp(OpcodeArgs);
void IMUL1SrcOp(OpcodeArgs);
void IMUL2SrcOp(OpcodeArgs);
void IMULOp(OpcodeArgs);
@@ -924,36 +916,17 @@ public:
}
protected:
void SaveNZCV(IROps Op = OP_DUMMY) override {
/* Some opcodes are conservatively marked as clobbering flags, but in fact
* do not clobber flags in certain conditions. Check for that here as an
* optimization.
*/
switch (Op) {
case OP_VFMINSCALARINSERT:
case OP_VFMAXSCALARINSERT:
/* On AFP platforms, becomes fmin/fmax and preserves NZCV. Otherwise
* becomes fcmp and clobbers.
*/
if (CTX->HostFeatures.SupportsAFP)
return;
break;
default:
break;
}
// Invariant: When executing instructions that clobber NZCV, the flags must
// be resident in a GPR, which is equivalent to CachedNZCV != nullptr. Get
// the NZCV which fills the cache if necessary.
if (CachedNZCV == nullptr)
GetNZCV();
// Assume we'll need a reload.
NZCVDirty = true;
void SaveNZCV() override {
}
private:
enum class SelectionFlag {
Nothing, // must rely on x86 flags
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
FCMP, // flags were set by a ucomis* / comis*
};
struct JumpTargetInfo {
OrderedNode* BlockEntry;
bool HaveEmitted;
@@ -961,6 +934,13 @@ private:
FEXCore::Context::ContextImpl *CTX{};
SelectionFlag flagsOp{};
uint8_t flagsOpSize{};
OrderedNode* flagsOpDest{};
OrderedNode* flagsOpSrc{};
OrderedNode* flagsOpDestSigned{};
OrderedNode* flagsOpSrcSigned{};
constexpr static unsigned FullNZCVMask =
(1U << FEXCore::X86State::RFLAG_CF_RAW_LOC) |
(1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC) |
@@ -1263,7 +1243,7 @@ private:
OrderedNode *GetNZCV() {
if (!CachedNZCV) {
CachedNZCV = _LoadNZCV();
CachedNZCV = _LoadFlag(FEXCore::X86State::RFLAG_NZCV_LOC);
// We don't know what's set
PossiblySetNZCVBits = ~0;
@@ -1295,85 +1275,37 @@ private:
}
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
CachedNZCV = _LoadNZCV();
CachedNZCV = _TestNZ(SrcSize, Res);
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
NZCVDirty = false;
NZCVDirty = true;
}
void InsertNZCV(unsigned BitOffset, OrderedNode *Value, signed FlagOffset, bool MustMask) {
signed Bit = IndexNZCV(BitOffset);
// If NZCV is not dirty, we always want to use rmif, it's 1 instruction to
// implement this. But if NZCV is dirty, it might still be cheaper to copy
// the GPR flags to NZCV and rmif. This is a heuristic for cases where we
// expect that 2 instruction sequence to be a win (versus something like
// bfe+mov+bfi+mov which can happen with our RA..). It's not totally
// conservative but it's pretty good in practice.
bool PreferRmif = !NZCVDirty || FlagOffset || MustMask ||
(PossiblySetNZCVBits & (1u << Bit));
if (CTX->HostFeatures.SupportsFlagM && PreferRmif) {
// Update NZCV
if (NZCVDirty && CachedNZCV)
_StoreNZCV(CachedNZCV);
CachedNZCV = nullptr;
NZCVDirty = false;
// Insert as NZCV.
signed RmifBit = Bit - 28;
_RmifNZCV(Value, (64 + FlagOffset - RmifBit) % 64, 1u << RmifBit);
CachedNZCV = nullptr;
} else {
// Insert as GPR
if (FlagOffset || MustMask)
Value = _Bfe(OpSize::i64Bit, 1, FlagOffset, Value);
if (PossiblySetNZCVBits == 0)
SetNZCV(_Lshl(OpSize::i64Bit, Value, _Constant(Bit)));
else if ((PossiblySetNZCVBits & (1u << Bit)) == 0)
SetNZCV(_Orlshl(OpSize::i32Bit, GetNZCV(), Value, Bit));
else
SetNZCV(_Bfi(OpSize::i32Bit, 1, Bit, GetNZCV(), Value));
}
OrderedNode *InsertNZCV(OrderedNode *NZCV, unsigned BitOffset, OrderedNode *Value) {
unsigned Bit = IndexNZCV(BitOffset);
uint32_t SetBits = PossiblySetNZCVBits;
PossiblySetNZCVBits |= (1u << Bit);
}
void CarryInvert() {
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
if (CTX->HostFeatures.SupportsFlagM && !NZCVDirty) {
// Invert as NZCV.
_CarryInvert();
CachedNZCV = nullptr;
} else {
// Invert as a GPR
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), _Constant(1u << Bit)));
}
PossiblySetNZCVBits |= 1u << Bit;
if (SetBits == 0)
return _Lshl(OpSize::i64Bit, Value, _Constant(Bit));
else if ((SetBits & (1u << Bit)) == 0)
return _Orlshl(OpSize::i32Bit, NZCV, Value, Bit);
else
return _Bfi(OpSize::i32Bit, 1, Bit, NZCV, Value);
}
template<unsigned BitOffset>
void SetRFLAG(OrderedNode *Value, unsigned ValueOffset = 0, bool MustMask = false) {
SetRFLAG(Value, BitOffset, ValueOffset, MustMask);
void SetRFLAG(OrderedNode *Value) {
SetRFLAG(Value, BitOffset);
}
void SetRFLAG(OrderedNode *Value, unsigned BitOffset, unsigned ValueOffset = 0, bool MustMask = false) {
if (IsNZCV(BitOffset)) {
InsertNZCV(BitOffset, Value, ValueOffset, MustMask);
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
_StoreRegister(Value, false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
_StoreRegister(Value, false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
} else {
if (ValueOffset || MustMask)
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
void SetRFLAG(OrderedNode *Value, unsigned BitOffset) {
flagsOp = SelectionFlag::Nothing;
if (IsNZCV(BitOffset))
SetNZCV(InsertNZCV(PossiblySetNZCVBits ? GetNZCV() : nullptr, BitOffset, Value));
else
_StoreFlag(Value, BitOffset);
}
}
void SetAF(unsigned Constant) {
@@ -1386,134 +1318,17 @@ private:
void ZeroMultipleFlags(uint32_t BitMask);
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
switch (BitOffset) {
case FEXCore::X86State::RFLAG_SF_RAW_LOC:
return Invert ? CondClassType{COND_PL} : CondClassType{COND_MI};
case FEXCore::X86State::RFLAG_ZF_RAW_LOC:
return Invert ? CondClassType{COND_NEQ} : CondClassType{COND_EQ};
case FEXCore::X86State::RFLAG_CF_RAW_LOC:
return Invert ? CondClassType{COND_ULT} : CondClassType{COND_UGE};
case FEXCore::X86State::RFLAG_OF_RAW_LOC:
return Invert ? CondClassType{COND_FNU} : CondClassType{COND_FU};
default:
FEX_UNREACHABLE;
}
}
OrderedNode *GetRFLAG(unsigned BitOffset, bool Invert = false) {
OrderedNode *GetRFLAG(unsigned BitOffset) {
if (IsNZCV(BitOffset)) {
if (!(PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset)))) {
return _Constant(Invert ? 1 : 0);
} else if (NZCVDirty) {
auto Value = _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV());
if (Invert)
return _Xor(OpSize::i32Bit, Value, _Constant(1));
else
return Value;
} else {
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert),
_Constant(1), _Constant(0));
}
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
if (!CachedNZCV || (PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset))))
return _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV());
else
return _Constant(0);
} else {
return _LoadFlag(BitOffset);
}
}
// Set SSE comparison flags based on the result set by Arm FCMP. This converts
// NZCV from the Arm representation to an eXternal representation that's
// totally not a euphemism for x86 or anything, nuh-uh.
void ConvertNZCVToSSE() {
if (CTX->HostFeatures.SupportsFlagM2) {
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
// We need to set PF according to the unordered flag. We'd rather do this
// after axflag, since some impls fuse fcmp+axflag, so we want to do this
// after. We can recover "unordered" after axflag as (Z && !C), but
// there's no condition code for this so it would take 2 instructions
// instead of one, which seems worse than doing 1 op before and breaking
// the fusion.
//
// We set PF to unordered (V), but our PF representation is inverted so we
// actually set to !V. This is one instruction with the VC cond code.
OrderedNode *PFInvert =
_NZCVSelect(OpSize::i32Bit, CondClassType{COND_FNU}, _Constant(1), _Constant(0));
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PFInvert);
// For the rest, this one weird a64 instruction maps exactly to what x86
// needs. What a coincidence!
_AXFlag();
PossiblySetNZCVBits = ~0;
// It does assume we invert CF internally, which is still TODO for us. For
// now, add a cfinv to deal. Hopefully we delete this later.
CarryInvert();
} else {
OrderedNode *Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
OrderedNode *C_inv = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
OrderedNode *V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
// We want to zero SF/OF, and then set CF/ZF. Zeroing up front lets us do
// this all with shifted-or's on non-flagm platforms.
ZeroNZCV();
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Or(OpSize::i32Bit, C_inv, V));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Or(OpSize::i32Bit, Z, V));
// Note that we store PF inverted.
// TODO: We could maybe optimize this xor out for non-flagm platforms with
// bfi/bfxil?
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Xor(OpSize::i32Bit, V, _Constant(1)));
}
}
// Set x87 comparison flags based on the result set by Arm FCMP. Clobbers
// NZCV on flagm2 platforms.
void ConvertNZCVToX87() {
OrderedNode *V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
if (CTX->HostFeatures.SupportsFlagM2) {
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
// Convert to x86 flags, saves us from or'ing after.
_AXFlag();
PossiblySetNZCVBits = ~0;
// Copy the values. CF is inverted from the axflag result, ZF is as-is.
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true));
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC));
} else {
OrderedNode *Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
OrderedNode *N = GetRFLAG(FEXCore::X86State::RFLAG_SF_RAW_LOC);
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Or(OpSize::i32Bit, N, V));
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Or(OpSize::i32Bit, Z, V));
}
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(V);
}
// Helper to derive Dest by a given builder-using Expression with the opcode
// replaced with NewOp. Useful for generic building code. Not safe in general.
// but does the right handling of ImplicitFlagClobber at least and must be
// used instead of raw Op mutation.
#define DeriveOp(Dest, NewOp, Expr) \
if (ImplicitFlagClobber(NewOp)) \
SaveNZCV(NewOp); \
auto Dest = (Expr); \
Dest.first->Header.Op = (NewOp)
// Named constant cache for the current block.
// Different arrays for sizes 1,2,4,8,16,32.
OrderedNode *CachedNamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_MAX][6]{};
@@ -1567,8 +1382,8 @@ private:
CachedIndexedNamedVectorConstants.clear();
}
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP) const;
OrderedNode *SelectBit(OrderedNode *Cmp, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
OrderedNode *SelectMask(OrderedNode *Cmp, uint64_t Mask, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
OrderedNode *SelectNZCV(unsigned BitOffset, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
OrderedNode *SelectCC(uint8_t OP, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
/**
@@ -1601,7 +1416,7 @@ private:
OrderedNode *Res{};
union {
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, RDRAND
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, BITSELECT, RDRAND
struct {
} NoSource;
@@ -1671,48 +1486,13 @@ private:
return CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE;
}
template <typename F>
void CalculateFlags_ShiftVariable(OrderedNode *Shift, F&& CalculateFlags) {
// We are the ones calculating the deferred flags. Don't recurse!
InvalidateDeferredFlags();
// RCR can call this with constants, so handle that without branching.
uint64_t Const;
if (IsValueConstant(WrapNode(Shift), &Const)) {
if (Const)
CalculateFlags();
return;
}
// Otherwise, prepare to branch.
uint32_t OldSetNZCVBits = PossiblySetNZCVBits;
auto Zero = _Constant(0);
// If the shift is zero, do not touch the flags.
auto SetBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
auto EndBlock = CreateNewCodeBlockAfter(SetBlock);
CondJump(Shift, Zero, EndBlock, SetBlock, {COND_EQ});
SetCurrentCodeBlock(SetBlock);
StartNewBlock();
{
CalculateFlags();
Jump(EndBlock);
}
SetCurrentCodeBlock(EndBlock);
StartNewBlock();
PossiblySetNZCVBits |= OldSetNZCVBits;
}
/**
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
* @{ */
OrderedNode *LoadPFRaw();
OrderedNode *LoadAF();
void FixupAF();
void CalculatePF(OrderedNode *Res);
void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr);
void CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool Sub);
@@ -1735,6 +1515,7 @@ private:
void CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateFlags_BEXTR(OrderedNode *Src);
void CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
void CalculateFlags_BLSMSK(OrderedNode *Src);
@@ -1743,6 +1524,7 @@ private:
void CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
void CalculateFlags_TZCNT(OrderedNode *Src);
void CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
void CalculateFlags_BITSELECT(OrderedNode *Src);
void CalculateFlags_RDRAND(OrderedNode *Src);
/** @} */
@@ -2045,6 +1827,20 @@ private:
};
}
void GenerateFlags_FCMP(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_FCMP,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.TwoSource = {
.Src1 = Src1,
.Src2 = Src2,
},
}
};
}
void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_BEXTR,
@@ -2119,6 +1915,14 @@ private:
};
}
void GenerateFlags_BITSELECT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_BITSELECT,
.SrcSize = GetSrcSize(Op),
.Res = Src,
};
}
void GenerateFlags_RDRAND(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_RDRAND,
@@ -26,24 +26,9 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *RotatedNode{};
if (CTX->HostFeatures.SupportsSHA) {
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
}
else {
// SHA1 extension missing, manually rotate.
// Emulate rotate.
auto ShiftLeft = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Dest, 30);
RotatedNode = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeft, Dest, 2);
}
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
auto Tmp = _Ror(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), _Constant(32, 2));
auto Top = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Src, 3), Tmp);
auto Result = _VInsGPR(16, 4, 3, Src, Top);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -64,31 +49,23 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
// Do all the work without it.
// ROR by 31 is equivalent to a ROL by 1
auto ThirtyOne = _Constant(32, 31);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(OpSize::i32Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
auto W13 = _VExtractToGPR(16, 4, Src, 2);
auto W14 = _VExtractToGPR(16, 4, Src, 1);
auto W15 = _VExtractToGPR(16, 4, Src, 0);
auto W16 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), W13), ThirtyOne);
auto W17 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 2), W14), ThirtyOne);
auto W18 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 1), W15), ThirtyOne);
auto W19 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 0), W16), ThirtyOne);
// Shift the incoming source left by a 32-bit element, inserting Zeros.
// This could be slightly improved to use a VInsGPR with the zero register.
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
auto D3 = _VInsGPR(16, 4, 3, Dest, W16);
auto D2 = _VInsGPR(16, 4, 2, D3, W17);
auto D1 = _VInsGPR(16, 4, 1, D2, W18);
auto D0 = _VInsGPR(16, 4, 0, D1, W19);
// Emulate rotate.
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
// Element0 didn't get XOR'd with anything, so do it now.
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
// Emulate rotate.
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
StoreResult(FPRClass, Op, Result, -1);
StoreResult(FPRClass, Op, D0, -1);
}
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
@@ -174,37 +151,30 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
}
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
};
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Result{};
auto W4 = _VExtractToGPR(16, 4, Src, 0);
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
if (CTX->HostFeatures.SupportsSHA) {
Result = _VSha256U0(Dest, Src);
}
else {
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
};
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
auto W4 = _VExtractToGPR(16, 4, Src, 0);
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
auto D0 = _VInsGPR(16, 4, 0, D1, Sig0);
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
Result = _VInsGPR(16, 4, 0, D1, Sig0);
}
StoreResult(FPRClass, Op, Result, -1);
StoreResult(FPRClass, Op, D0, -1);
}
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
@@ -40,6 +40,7 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
};
void OpDispatchBuilder::ZeroMultipleFlags(uint32_t FlagsMask) {
flagsOp = SelectionFlag::Nothing;
auto ZeroConst = _Constant(0);
if (ContainsNZCV(FlagsMask)) {
@@ -129,7 +130,8 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
Tmp = _Xor(OpSize::i32Bit, Tmp, _Constant(1));
SetRFLAG(Tmp, FlagOffset);
} else {
SetRFLAG(Src, FlagOffset, FlagOffset, true);
auto Tmp = _Bfe(OpSize::i32Bit, 1, FlagOffset, Src);
SetRFLAG(Tmp, FlagOffset);
}
}
}
@@ -235,17 +237,17 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNo
Anded = _Andn(OpSize, XorOp2, XorOp1);
}
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
auto OF = _Bfe(OpSize, 1, SrcSize * 8 - 1, Anded);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OF);
}
OrderedNode *OpDispatchBuilder::LoadPFRaw() {
// Read the stored byte. This is the original result (up to 64-bits), it needs
// parity calculated.
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
// Read the stored byte. This is the original 8-bit result, it needs parity calculated.
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
// Cast the input to a 32-bit FPR. Logically we only need 8-bit, but that would
// generate unwanted an ubfx instruction. VPopcount will ignore the upper bits anyway.
auto InputFPR = _VCastFromGPR(4, 4, Result);
auto InputFPR = _VCastFromGPR(4, 4, PFByte);
// Calculate the popcount.
auto Count = _VPopcount(1, 1, InputFPR);
@@ -253,14 +255,14 @@ OrderedNode *OpDispatchBuilder::LoadPFRaw() {
}
OrderedNode *OpDispatchBuilder::LoadAF() {
// Read the stored value. This is the XOR of the arguments.
auto AFWord = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
// Read the stored byte. This is the XOR of the arguments.
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
// Read the result, stored for PF.
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
// Read the result, stored as the PF byte for deferred PF calculation.
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
// What's left is to XOR and extract. This is the deferred part.
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i32Bit, AFWord, Result));
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i32Bit, AFByte, PFByte));
}
void OpDispatchBuilder::FixupAF() {
@@ -270,14 +272,23 @@ void OpDispatchBuilder::FixupAF() {
//
// (AF[4] ^ PF[4]) ^ PF[4] = AF[4]
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
OrderedNode *XorRes = _Xor(OpSize::i32Bit, AFRaw, PFRaw);
OrderedNode *XorRes = _Xor(OpSize::i32Bit, AFByte, PFByte);
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
}
void OpDispatchBuilder::CalculatePF(OrderedNode *Res) {
void OpDispatchBuilder::CalculatePF(OrderedNode *Res, OrderedNode *condition) {
// For shifts, we can only update for nonzero shift. If zero, we nop out the flag write by
// writing the existing value. Note we call GetRFLAG directly, rather than LoadPFRaw, because
// we need the existing /encoded/ value rather than the decoded PF value. In particular,
// this does not calculate a popcount.
if (condition) {
auto OldFlag = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
Res = _Select(FEXCore::IR::COND_EQ, condition, _Constant(0), OldFlag, Res);
}
// Calculation is entirely deferred until load, just store the 8-bit result.
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Res);
}
@@ -302,7 +313,7 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
if (CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE) {
// Nothing to do
if (NZCVDirty && CachedNZCV)
_StoreNZCV(CachedNZCV);
_StoreFlag(CachedNZCV, FEXCore::X86State::RFLAG_NZCV_LOC);
CachedNZCV = nullptr;
NZCVDirty = false;
@@ -435,6 +446,13 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
break;
case FlagsGenerationType::TYPE_FCMP:
CalculateFlags_FCMP(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.TwoSource.Src1,
CurrentDeferredFlags.Sources.TwoSource.Src2);
break;
case FlagsGenerationType::TYPE_BEXTR:
CalculateFlags_BEXTR(CurrentDeferredFlags.Res);
break;
@@ -469,6 +487,9 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res);
break;
case FlagsGenerationType::TYPE_BITSELECT:
CalculateFlags_BITSELECT(CurrentDeferredFlags.Res);
break;
case FlagsGenerationType::TYPE_RDRAND:
CalculateFlags_RDRAND(CurrentDeferredFlags.Res);
break;
@@ -480,7 +501,7 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
CurrentDeferredFlags.Type = FlagsGenerationType::TYPE_NONE;
if (NZCVDirty && CachedNZCV)
_StoreNZCV(CachedNZCV);
_StoreFlag(CachedNZCV, FEXCore::X86State::RFLAG_NZCV_LOC);
CachedNZCV = nullptr;
NZCVDirty = false;
@@ -495,12 +516,7 @@ void OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, Or
CalculatePF(Res);
if (SrcSize >= 4) {
if (NZCVDirty && CachedNZCV)
_StoreNZCV(CachedNZCV);
CachedNZCV = nullptr;
_AdcNZCV(OpSize, Src1, Src2);
PossiblySetNZCVBits = ~0;
SetNZCV(_AdcNZCV(OpSize, Src1, Src2, GetNZCV()));
} else {
// SF/ZF
SetNZ_ZeroCV(SrcSize, Res);
@@ -528,19 +544,7 @@ void OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, Or
CalculatePF(Res);
if (SrcSize >= 4) {
// Rectify input carry
CarryInvert();
if (NZCVDirty && CachedNZCV)
_StoreNZCV(CachedNZCV);
CachedNZCV = nullptr;
NZCVDirty = false;
_SbbNZCV(OpSize, Src1, Src2);
PossiblySetNZCVBits = ~0;
// Rectify output carry
CarryInvert();
SetNZCV(_SbbNZCV(OpSize, Src1, Src2, GetNZCV()));
} else {
// SF/ZF
SetNZ_ZeroCV(SrcSize, Res);
@@ -570,14 +574,8 @@ void OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, Or
// TODO: Could do this path for small sources if we have FEAT_FlagM
if (SrcSize >= 4) {
_SubNZCV(OpSize, Src1, Src2);
CachedNZCV = nullptr;
NZCVDirty = false;
PossiblySetNZCVBits = ~0;
// We only bother inverting CF if we're actually going to update CF.
if (UpdateCF)
CarryInvert();
SetNZCV(_SubNZCV(OpSize, Src1, Src2, UpdateCF));
} else {
// SF/ZF
SetNZ_ZeroCV(SrcSize, Res);
@@ -585,7 +583,8 @@ void OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, Or
// CF
if (UpdateCF) {
// Grab carry bit from unmasked output.
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SrcSize * 8, true);
auto Bfe = _Bfe(OpSize::i32Bit, 1, SrcSize * 8, Res);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Bfe);
}
CalculateOF(SrcSize, Res, Src1, Src2, true);
@@ -607,10 +606,7 @@ void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, Or
// TODO: Could do this path for small sources if we have FEAT_FlagM
if (SrcSize >= 4) {
_AddNZCV(OpSize, Src1, Src2);
CachedNZCV = nullptr;
NZCVDirty = false;
PossiblySetNZCVBits = ~0;
SetNZCV(_AddNZCV(OpSize, Src1, Src2));
} else {
// SF/ZF
SetNZ_ZeroCV(SrcSize, Res);
@@ -618,7 +614,8 @@ void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, Or
// CF
if (UpdateCF) {
// Grab carry bit from unmasked output
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SrcSize * 8, true);
auto Bfe = _Bfe(OpSize::i32Bit, 1, SrcSize * 8, Res);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Bfe);
}
CalculateOF(SrcSize, Res, Src1, Src2, false);
@@ -630,6 +627,8 @@ void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, Or
}
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High) {
auto Zero = _Constant(0);
// PF/AF/ZF/SF
// Undefined
{
@@ -641,22 +640,19 @@ void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, Or
{
// CF and OF are set if the result of the operation can't be fit in to the destination register
// If the value can fit then the top bits will be zero
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
_SubNZCV(OpSize::i64Bit, High, SignBit);
// If High = SignBit, then sets to nZcv. Else sets to nzCV. Since SF/ZF
// undefined, this does what we need.
auto Zero = _Constant(0);
_CondAddNZCV(OpSize::i64Bit, Zero, Zero, CondClassType{COND_EQ}, 0x3 /* nzCV */);
CachedNZCV = nullptr;
NZCVDirty = false;
PossiblySetNZCVBits = ~0;
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
auto CV = _Constant((1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC)) |
(1u << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC)));
// Set CV accordingly and zero NZ regardless
SetNZCV(_Select(FEXCore::IR::COND_EQ, High, SignBit, Zero, CV));
}
}
void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
auto Zero = _Constant(0);
OpSize Size = IR::SizeToOpSize(GetOpSize(High));
// AF/SF/PF/ZF
// Undefined
@@ -669,14 +665,11 @@ void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
{
// CF and OF are set if the result of the operation can't be fit in to the destination register
// The result register will be all zero if it can't fit due to how multiplication behaves
_SubNZCV(Size, High, Zero);
// If High = 0, then sets to nZcv. Else sets to nzCV. Since SF/ZF undefined,
// this does what we need.
_CondAddNZCV(Size, Zero, Zero, CondClassType{COND_EQ}, 0x3 /* nzCV */);
CachedNZCV = nullptr;
NZCVDirty = false;
PossiblySetNZCVBits = ~0;
auto CV = _Constant((1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC)) |
(1u << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC)));
SetNZCV(_Select(FEXCore::IR::COND_EQ, High, Zero, Zero, CV));
}
}
@@ -692,79 +685,118 @@ void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, OrderedNode *Res
}
void OpDispatchBuilder::CalculateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res, Src1, Src2](){
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
SetNZ_ZeroCV(SrcSize, Res);
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto Zero = _Constant(0);
auto OldNZCV = GetNZCV();
uint32_t OldSetNZCVBits = PossiblySetNZCVBits;
SetNZ_ZeroCV(SrcSize, Res);
// CF
{
// Extract the last bit shifted in to CF
auto Size = _Constant(SrcSize * 8);
auto ShiftAmt = _Sub(OpSize, Size, Src2);
auto LastBit = _Lshr(OpSize, Src1, ShiftAmt);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit, 0, true);
auto LastBit = _Bfe(OpSize, 1, 0, _Lshr(OpSize, Src1, ShiftAmt));
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit);
}
CalculatePF(Res);
CalculatePF(Res, Src2);
// AF
// Undefined
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
// AF
// Undefined
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
// OF
{
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
// When Shift > 1 then OF is undefined
auto OFXor = _Xor(OpSize, Src1, Res);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OFXor, SrcSize * 8 - 1, true);
});
auto val = _Bfe(OpSize, 1, SrcSize * 8 - 1, _Xor(OpSize, Src1, Res));
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val);
}
// Now select between the two
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
PossiblySetNZCVBits |= OldSetNZCVBits;
}
void OpDispatchBuilder::CalculateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res, Src1, Src2](){
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
SetNZ_ZeroCV(SrcSize, Res);
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto Zero = _Constant(0);
auto One = _Constant(1);
auto OldNZCV = GetNZCV();
uint32_t OldSetNZCVBits = PossiblySetNZCVBits;
SetNZ_ZeroCV(SrcSize, Res);
// CF
{
// Extract the last bit shifted in to CF
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, _Constant(1));
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, One);
const auto CFSize = IR::SizeToOpSize(std::max<uint8_t>(4u, SrcSize));
auto LastBit = _Lshr(CFSize, Src1, ShiftAmt);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit, 0, true);
auto LastBit = _Bfe(CFSize, 1, 0, _Lshr(CFSize, Src1, ShiftAmt));
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit);
}
CalculatePF(Res);
CalculatePF(Res, Src2);
// AF
// Undefined
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
// AF
// Undefined
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
// OF
{
// Only defined when Shift is 1 else undefined
// OF flag is set if a sign change occurred
auto val = _Xor(OpSize, Src1, Res);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val, SrcSize * 8 - 1, true);
});
auto val = _Bfe(OpSize, 1, SrcSize * 8 - 1, _Xor(OpSize, Src1, Res));
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val);
}
// Now select between the two
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
PossiblySetNZCVBits |= OldSetNZCVBits;
}
void OpDispatchBuilder::CalculateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res, Src1, Src2](){
// SF/ZF/OF
SetNZ_ZeroCV(SrcSize, Res);
auto Zero = _Constant(0);
auto One = _Constant(1);
auto OldNZCV = GetNZCV();
uint32_t OldSetNZCVBits = PossiblySetNZCVBits;
// SF/ZF/OF
SetNZ_ZeroCV(SrcSize, Res);
// CF
{
// Extract the last bit shifted in to CF
const auto CFSize = IR::SizeToOpSize(std::max<uint32_t>(4u, GetOpSize(Src1)));
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, _Constant(1));
auto LastBit = _Lshr(CFSize, Src1, ShiftAmt);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit, 0, true);
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, One);
auto LastBit = _Bfe(CFSize, 1, 0, _Lshr(CFSize, Src1, ShiftAmt));
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit);
}
CalculatePF(Res);
CalculatePF(Res, Src2);
// AF
// Undefined
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
});
// AF
// Undefined
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
// Now select between the two
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
PossiblySetNZCVBits |= OldSetNZCVBits;
}
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *UnmaskedRes, OrderedNode *Src1, uint64_t Shift) {
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
// No flags changed if shift is zero
if (Shift == 0) return;
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
SetNZ_ZeroCV(SrcSize, UnmaskedRes);
SetNZ_ZeroCV(SrcSize, Res);
// CF
{
@@ -773,10 +805,10 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Order
if (SrcSizeBits < Shift) {
Shift &= (SrcSizeBits - 1);
}
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, SrcSizeBits - Shift, true);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Bfe(OpSize, 1, SrcSizeBits - Shift, Src1));
}
CalculatePF(UnmaskedRes);
CalculatePF(Res);
// AF
// Undefined
@@ -785,8 +817,9 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Order
// OF
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
if (Shift == 1) {
auto Xor = _Xor(OpSize, UnmaskedRes, Src1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, SrcSize * 8 - 1, true);
auto Xor = _Xor(OpSize, Res, Src1);
auto OF = _Bfe(OpSize, 1, SrcSize * 8 - 1, Xor);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OF);
} else {
// Undefined, we choose to zero as part of SetNZ_ZeroCV
}
@@ -801,7 +834,7 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift-1, true);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Bfe(IR::SizeToOpSize(std::max<uint32_t>(4u, GetOpSize(Src1))), 1, Shift-1, Src1));
}
CalculatePF(Res);
@@ -817,6 +850,8 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
}
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
// Stash OF before overwriting it
auto OldOF = Shift != 1 ? GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC) : NULL;
SetNZ_ZeroCV(SrcSize, Res);
@@ -824,7 +859,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift-1, true);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Bfe(OpSize, 1, Shift-1, Src1));
}
CalculatePF(Res);
@@ -843,6 +878,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Orde
// No flags changed if shift is zero
if (Shift == 0) return;
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
CalculateFlags_ShiftRightImmediateCommon(SrcSize, Res, Src1, Shift);
// OF
@@ -850,7 +886,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Orde
// Only defined when Shift is 1 else undefined
// Is set to the MSB of the original value
if (Shift == 1) {
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Src1, SrcSize * 8 - 1, true);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(_Bfe(OpSize, 1, SrcSize * 8 - 1, Src1));
}
}
}
@@ -868,53 +904,62 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
// Is set if the MSB bit changes.
// XOR of Result and Src1
if (Shift == 1) {
auto val = _Xor(OpSize, Src1, Res);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val, SrcSize * 8 - 1, true);
auto val = _Bfe(OpSize, 1, SrcSize * 8 - 1, _Xor(OpSize, Src1, Res));
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val);
}
}
}
void OpDispatchBuilder::CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
auto SizeBits = SrcSize * 8;
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto Zero = _Constant(0);
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
auto OldNZCV = GetNZCV();
auto OldSetNZCVBits = PossiblySetNZCVBits;
ZeroCV();
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
// Extract the last bit shifted in to CF
auto NewCF = _Bfe(OpSize, 1, SizeBits - 1, Res);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
// OF is set to the XOR of the new CF bit and the most significant bit of the result
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, true);
});
// OF is set to the XOR of the new CF bit and the most significant bit of the result
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 2, Res), NewCF);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
// Now select: if shift == 0, don't update flags
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
PossiblySetNZCVBits |= OldSetNZCVBits;
}
void OpDispatchBuilder::CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
auto Zero = _Constant(0);
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
auto OldNZCV = GetNZCV();
auto OldSetNZCVBits = PossiblySetNZCVBits;
// Extract the last bit shifted in to CF
//auto Size = _Constant(GetSrcSize(Res) * 8);
//auto ShiftAmt = _Sub(OpSize::i64Bit, Size, Src2);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
// Ends up faster overall.
// XXX: can do much better if we have FlagM (with RMIF).
ZeroCV();
// OF is the LSB and MSB XOR'd together.
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
});
// Extract the last bit shifted in to CF
//auto Size = _Constant(GetSrcSize(Res) * 8);
//auto ShiftAmt = _Sub(OpSize::i64Bit, Size, Src2);
auto NewCF = _Bfe(OpSize, 1, 0, Res);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
// OF is the LSB and MSB XOR'd together.
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 1, Res), NewCF);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
// Now select: if shift == 0, don't update flags
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
PossiblySetNZCVBits |= OldSetNZCVBits;
}
void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
@@ -922,16 +967,16 @@ void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, Ord
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
auto NewCF = _Bfe(OpSize, 1, SizeBits - 1, Res);
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
// Ends up faster overall. If Shift != 1, OF is undefined so we choose to zero here.
// XXX: can do much better if we have FlagM (with RMIF).
ZeroCV();
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
}
// OF
@@ -939,8 +984,8 @@ void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, Ord
if (Shift == 1) {
// OF is the top two MSBs XOR'd together
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, 1);
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 2, Res), NewCF);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
}
}
}
@@ -951,15 +996,16 @@ void OpDispatchBuilder::CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, Orde
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
auto NewCF = _Bfe(OpSize, 1, 0, Res);
// Ends up faster overall. If Shift != 1, OF is undefined so we choose to zero here.
// XXX: can do much better if we have FlagM (with RMIF).
ZeroCV();
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
}
// OF
@@ -968,13 +1014,37 @@ void OpDispatchBuilder::CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, Orde
// OF is the LSB and MSB XOR'd together.
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 1, Res), NewCF);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
}
}
}
void OpDispatchBuilder::CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
// PF is stored inverted, so invert from the host flag.
// TODO: This could perhaps be optimized?
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
// Zero AF. Note that we set the PF byte to 0/1 above, so PF[4] is 0 so the
// XOR with PF will have no effect, so setting the AF byte to zero will indeed
// zero AF as intended.
uint32_t FlagsMaskToZero =
(1U << X86State::RFLAG_AF_RAW_LOC) |
(1U << X86State::RFLAG_SF_RAW_LOC) |
(1U << X86State::RFLAG_OF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
}
void OpDispatchBuilder::CalculateFlags_BEXTR(OrderedNode *Src) {
auto Zero = _Constant(0);
auto One = _Constant(1);
@@ -1085,12 +1155,34 @@ void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode *Src) {
}
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
// Now for the flags
auto Bounds = _Constant(SrcSize * 8- 1);
auto Zero = _Constant(0);
auto One = _Constant(1);
// OF cleared
SetRFLAG<X86State::RFLAG_OF_RAW_LOC>(Zero);
// PF/AF undefined
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
(1UL << X86State::RFLAG_AF_RAW_LOC));
SetNZ_ZeroCV(SrcSize, Result);
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(Src);
// ZF
{
auto ZFOp = _Select(IR::COND_EQ,
Result, Zero,
One, Zero);
SetRFLAG<X86State::RFLAG_ZF_RAW_LOC>(ZFOp);
}
// CF
{
auto CFOp = _Select(IR::COND_UGT,
Src, Bounds,
One, Zero);
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
}
}
void OpDispatchBuilder::CalculateFlags_TZCNT(OrderedNode *Src) {
@@ -1104,10 +1196,12 @@ void OpDispatchBuilder::CalculateFlags_TZCNT(OrderedNode *Src) {
// Set flags
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(ZFResult);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Src, 0, true);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Bfe(OpSize::i32Bit, 1, 0, Src));
}
void OpDispatchBuilder::CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src) {
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
// OF, SF, AF, PF all undefined
ZeroNZCV();
@@ -1118,7 +1212,22 @@ void OpDispatchBuilder::CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src)
// Set flags
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(ZFResult);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Src, SrcSize * 8 - 1, true);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Bfe(OpSize, 1, SrcSize * 8 - 1, Src));
}
void OpDispatchBuilder::CalculateFlags_BITSELECT(OrderedNode *Src) {
// OF, SF, AF, PF, CF all undefined
ZeroNZCV();
auto ZeroConst = _Constant(0);
auto OneConst = _Constant(1);
// ZF is set to 1 if the source was zero
auto ZFSelectOp = _Select(FEXCore::IR::COND_EQ,
Src, ZeroConst,
OneConst, ZeroConst);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(ZFSelectOp);
}
void OpDispatchBuilder::CalculateFlags_RDRAND(OrderedNode *Src) {
@@ -225,7 +225,9 @@ void OpDispatchBuilder::VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSi
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
DeriveOp(ALUOp, IROp, _VAdd(Size, ElementSize, Dest, Src));
auto ALUOp = _VAdd(Size, ElementSize, Dest, Src);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
StoreResult(FPRClass, Op, ALUOp, -1);
}
@@ -369,7 +371,9 @@ void OpDispatchBuilder::AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t Elemen
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
DeriveOp(ALUOp, IROp, _VAdd(Size, ElementSize, Src1, Src2));
auto ALUOp = _VAdd(Size, ElementSize, Src1, Src2);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
StoreResult(FPRClass, Op, ALUOp, -1);
}
@@ -502,7 +506,9 @@ void OpDispatchBuilder::VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementS
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
DeriveOp(ALUOp, IROp, _VAdd(Size, ElementSize, Src, Dest));
auto ALUOp = _VAdd(Size, ElementSize, Src, Dest);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
StoreResult(FPRClass, Op, ALUOp, -1);
}
@@ -534,8 +540,10 @@ OrderedNode* OpDispatchBuilder::VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IR
{.AllowUpperGarbage = true});
// If OpSize == ElementSize then it only does the lower scalar op
DeriveOp(ALUOp, IROp,
_VFAddScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits));
auto ALUOp = _VFAddScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
return ALUOp;
}
@@ -618,7 +626,10 @@ OrderedNode* OpDispatchBuilder::VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IRO
{.AllowUpperGarbage = true});
// If OpSize == ElementSize then it only does the lower scalar op
DeriveOp(ALUOp, IROp, _VFSqrtScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits));
auto ALUOp = _VFSqrtScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
return ALUOp;
}
@@ -929,7 +940,9 @@ void OpDispatchBuilder::VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t Element
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
DeriveOp(ALUOp, IROp, _VFSqrt(OpSize, ElementSize, Src));
auto ALUOp = _VFSqrt(OpSize, ElementSize, Src);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
StoreResult(FPRClass, Op, ALUOp, -1);
}
@@ -966,7 +979,9 @@ void OpDispatchBuilder::AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t Elem
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
DeriveOp(ALUOp, IROp, _VFSqrt(OpSize, ElementSize, Src));
auto ALUOp = _VFSqrt(OpSize, ElementSize, Src);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
// NOTE: We don't need to clear the upper lanes here, since the
// IR ops make use of 128-bit AdvSimd for 128-bit cases,
@@ -1002,7 +1017,9 @@ void OpDispatchBuilder::VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
DeriveOp(ALUOp, IROp, _VFSqrt(ElementSize, ElementSize, Src));
auto ALUOp = _VFSqrt(ElementSize, ElementSize, Src);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
// Duplicate the lower bits
auto Result = _VDupElement(Size, ElementSize, ALUOp, 0);
@@ -1729,7 +1746,8 @@ void OpDispatchBuilder::VHADDPOp(OpcodeArgs) {
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
DeriveOp(Res, IROp, _VFAddP(SrcSize, ElementSize, Src1, Src2));
auto Res = _VFAddP(SrcSize, ElementSize, Src1, Src2);
Res.first->Header.Op = IROp;
OrderedNode *Dest = Res;
if (Is256Bit) {
@@ -2421,7 +2439,8 @@ void OpDispatchBuilder::AVXVariableShiftImpl(OpcodeArgs, IROps IROp) {
OrderedNode *Vector = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags);
OrderedNode *ShiftVector = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], DstSize, Op->Flags);
DeriveOp(Shift, IROp, _VUShr(DstSize, SrcSize, Vector, ShiftVector, true));
auto Shift = _VUShr(DstSize, SrcSize, Vector, ShiftVector, true);
Shift.first->Header.Op = IROp;
StoreResult(FPRClass, Op, Shift, -1);
}
@@ -3422,21 +3441,20 @@ void OpDispatchBuilder::VPALIGNROp(OpcodeArgs) {
template<size_t ElementSize>
void OpDispatchBuilder::UCOMISxOp(OpcodeArgs) {
InvalidateDeferredFlags();
const auto SrcSize = Op->Src[0].IsGPR() ? GetGuestVectorLength() : GetSrcSize(Op);
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetGuestVectorLength(), Op->Flags);
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
OrderedNode *Res = _FCmp(ElementSize, Src1, Src2,
(1 << FCMP_FLAG_EQ) |
(1 << FCMP_FLAG_LT) |
(1 << FCMP_FLAG_UNORDERED));
CachedNZCV = nullptr;
_FCmp(ElementSize, Src1, Src2);
PossiblySetNZCVBits = ~0;
ConvertNZCVToSSE();
GenerateFlags_FCMP(Op, Res, Src1, Src2);
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so PF[4] is
// 0 so the XOR with PF will have no effect, so setting the AF byte to zero
// will indeed zero AF as intended.
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(0));
flagsOp = SelectionFlag::FCMP;
flagsOpDest = Src1;
flagsOpSrc = Src2;
flagsOpSize = GetSrcSize(Op);
}
template
@@ -247,13 +247,10 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
data = _Sbfe(OpSize::i64Bit, read_width * 8, 0, data);
}
// We're about to clobber flags to grab the sign, so save NZCV.
SaveNZCV();
// Extract sign and make interger absolute
_SubNZCV(OpSize::i64Bit, data, zero);
auto sign = _NZCVSelect(OpSize::i64Bit, CondClassType{COND_SLT}, _Constant(0x8000), zero);
auto absolute = _Neg(OpSize::i64Bit, data, CondClassType{COND_MI});
auto sign = _Select(COND_SLT, data, zero, _Constant(0x8000), zero);
auto absolute = _Abs(OpSize::i64Bit, data);
// left justify the absolute interger
auto shift = _Sub(OpSize::i64Bit, _Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
@@ -859,7 +856,9 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
auto top = GetX87Top();
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
DeriveOp(result, IROp, _F80Round(a));
auto result = _F80Round(a);
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F80SIN ||
IROp == IR::OP_F80COS) {
@@ -890,7 +889,9 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
st1 = _LoadContextIndexed(st1, 16, MMBaseOffset(), 16, FPRClass);
DeriveOp(result, IROp, _F80Add(a, st1));
auto result = _F80Add(a, st1);
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F80FPREM ||
IROp == IR::OP_F80FPREM1) {
@@ -601,13 +601,21 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
auto low = _Constant(0);
OrderedNode *data = _VCastFromGPR(8, 8, low);
// We are going to clobber NZCV, make sure it's in a GPR first.
GetNZCV();
OrderedNode *Res = _FCmp(8, a, data,
(1 << FCMP_FLAG_EQ) |
(1 << FCMP_FLAG_LT) |
(1 << FCMP_FLAG_UNORDERED));
// Now we do our comparison.
_FCmp(8, a, data);
PossiblySetNZCVBits = ~0;
ConvertNZCVToX87();
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
}
//TODO: This should obey rounding mode
@@ -673,22 +681,36 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
if constexpr (whichflags == FCOMIFlags::FLAGS_X87) {
// We are going to clobber NZCV, make sure it's in a GPR first.
GetNZCV();
OrderedNode *Res = _FCmp(8, a, b,
(1 << FCMP_FLAG_EQ) |
(1 << FCMP_FLAG_LT) |
(1 << FCMP_FLAG_UNORDERED));
_FCmp(8, a, b);
PossiblySetNZCVBits = ~0;
ConvertNZCVToX87();
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
if constexpr (whichflags == FCOMIFlags::FLAGS_X87) {
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
}
else {
// Invalidate deferred flags early
// OF, SF, AF, PF all undefined
InvalidateDeferredFlags();
_FCmp(8, a, b);
PossiblySetNZCVBits = ~0;
ConvertNZCVToSSE();
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
// PF is stored inverted, so invert from the host flag.
// TODO: This could perhaps be optimized?
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
}
if constexpr (poptwice) {
@@ -745,7 +767,9 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
auto top = GetX87Top();
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
DeriveOp(result, IROp, _F64SIN(a));
auto result = _F64SIN(a);
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F64SIN ||
IROp == IR::OP_F64COS) {
@@ -775,7 +799,9 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
st1 = _LoadContextIndexed(st1, 8, MMBaseOffset(), 16, FPRClass);
DeriveOp(result, IROp, _F64ATAN(a, st1));
auto result = _F64ATAN(a, st1);
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F64FPREM ||
IROp == IR::OP_F64FPREM1) {
@@ -0,0 +1,41 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <unistd.h>
#include <signal.h>
namespace FEXCore {
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
SetHostSignalHandler(Signal, Func, Required);
FrontendRegisterHostSignalHandler(Signal, Func, Required);
}
void SignalDelegator::HandleSignal(int Signal, void *Info, void *UContext) {
// Let the host take first stab at handling the signal
auto Thread = GetTLSThread();
HostSignalHandler &Handler = HostHandlers[Signal];
if (!Thread) {
LogMan::Msg::AFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
}
else {
for (auto &Handler : Handler.Handlers) {
if (Handler(Thread, Signal, Info, UContext)) {
// If the host handler handled the fault then we can continue now
return;
}
}
if (Handler.FrontendHandler &&
Handler.FrontendHandler(Thread, Signal, Info, UContext)) {
return;
}
// Now let the frontend handle the signal
// It's clearly a guest signal and this ends up being an OS specific issue
HandleGuestSignal(Thread, Signal, Info, UContext);
}
}
}
@@ -0,0 +1,84 @@
// SPDX-License-Identifier: MIT
#ifndef NDEBUG
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Utils/LogManager.h>
#include <tuple>
namespace FEXCore::X86Tables::X86InstDebugInfo {
void InstallDebugInfo() {
const std::tuple<uint8_t, uint8_t, Flags> BaseOpTable[] = {
{0x50, 8, {FLAGS_MEM_ACCESS}},
{0x58, 8, {FLAGS_MEM_ACCESS}},
{0x68, 1, {FLAGS_MEM_ACCESS}},
{0x6A, 1, {FLAGS_MEM_ACCESS}},
{0xAA, 4, {FLAGS_MEM_ACCESS}},
{0xC8, 1, {FLAGS_MEM_ACCESS}},
{0xCC, 2, {FLAGS_DEBUG}},
{0xD7, 1, {FLAGS_MEM_ACCESS}},
{0xF1, 1, {FLAGS_DEBUG}},
{0xF4, 1, {FLAGS_DEBUG}},
};
const std::tuple<uint8_t, uint8_t, Flags> TwoByteOpTable[] = {
{0x0B, 1, {FLAGS_DEBUG}},
{0x19, 7, {FLAGS_DEBUG}},
{0x28, 2, {FLAGS_MEM_ALIGN_16}},
{0x31, 1, {FLAGS_DEBUG}},
{0xA2, 1, {FLAGS_DEBUG}},
{0xA3, 1, {FLAGS_MEM_ACCESS}},
{0xAB, 1, {FLAGS_MEM_ACCESS}},
{0xB3, 1, {FLAGS_MEM_ACCESS}},
{0xBB, 1, {FLAGS_MEM_ACCESS}},
{0xFF, 1, {FLAGS_DEBUG}},
};
const std::tuple<uint8_t, uint8_t, Flags> PrimaryGroupOpTable[] = {
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 2, {FLAGS_DIVIDE}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 6), 2, {FLAGS_DIVIDE}},
#undef OPD
};
const std::tuple<uint16_t, uint8_t, Flags> SecondaryExtensionOpTable[] = {
#define PF_NONE 0
#define PF_F3 1
#define PF_66 2
#define PF_F2 3
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
{OPD(TYPE_GROUP_15, PF_NONE, 2), 1, {FLAGS_DEBUG}},
{OPD(TYPE_GROUP_15, PF_NONE, 3), 1, {FLAGS_DEBUG}},
#undef PF_F3
#undef PF_66
#undef PF_F2
#undef OPD
};
auto GenerateDebugTable = [](auto& FinalTable, auto& LocalTable) {
for (auto Op : LocalTable) {
auto OpNum = std::get<0>(Op);
auto DebugInfo = std::get<2>(Op);
for (uint8_t i = 0; i < std::get<1>(Op); ++i) {
memcpy(&FinalTable[OpNum+i].DebugInfo, &DebugInfo, sizeof(X86InstDebugInfo::Flags));
}
}
};
GenerateDebugTable(BaseOps, BaseOpTable);
GenerateDebugTable(SecondBaseOps, TwoByteOpTable);
GenerateDebugTable(PrimaryInstGroupOps, PrimaryGroupOpTable);
GenerateDebugTable(SecondInstGroupOps, SecondaryExtensionOpTable);
}
}
#endif
@@ -44,6 +44,11 @@ void InitializeVEXTables();
void InitializeXOPTables();
void InitializeEVEXTables();
#ifndef NDEBUG
uint64_t Total{};
uint64_t NumInsts{};
#endif
void InitializeInfoTables(Context::OperatingMode Mode) {
InitializeBaseTables(Mode);
InitializeSecondaryTables(Mode);
@@ -57,6 +62,10 @@ void InitializeInfoTables(Context::OperatingMode Mode) {
InitializeVEXTables();
InitializeXOPTables();
InitializeEVEXTables();
#ifndef NDEBUG
X86InstDebugInfo::InstallDebugInfo();
#endif
}
}
@@ -100,10 +100,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
{0x6B, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT , 1, nullptr}},
// This should just throw a GP
{0x6C, 1, X86InstInfo{"INSB", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{0x6D, 1, X86InstInfo{"INSW", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{0x6C, 1, X86InstInfo{"INSB", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
{0x6D, 1, X86InstInfo{"INSW", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
{0x70, 1, X86InstInfo{"JO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
{0x71, 1, X86InstInfo{"JNO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
@@ -147,19 +147,19 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
{0xB0, 8, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_REX_IN_BYTE , 1, nullptr}},
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2, 4, nullptr}},
@@ -169,7 +169,7 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
{0xC8, 1, X86InstInfo{"ENTER", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 3, nullptr}},
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
{0xCA, 2, X86InstInfo{"RETF", TYPE_PRIV, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1, nullptr}},
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
@@ -192,8 +192,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
{0xEC, 2, X86InstInfo{"IN", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{0xEE, 2, X86InstInfo{"OUT", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{0xF5, 1, X86InstInfo{"CMC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0xF8, 1, X86InstInfo{"CLC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0xF9, 1, X86InstInfo{"STC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
@@ -183,41 +183,41 @@ void InitializeSecondaryGroupTables() {
{OPD(TYPE_GROUP_9, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
// GROUP 10
{OPD(TYPE_GROUP_10, PF_NONE, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_NONE, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F3, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_66, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
{OPD(TYPE_GROUP_10, PF_F2, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
// GROUP 12
{OPD(TYPE_GROUP_12, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
@@ -29,7 +29,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x09, 1, X86InstInfo{"WBINVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x0A, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x0C, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x0D, 1, X86InstInfo{"", TYPE_GROUP_P, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x0E, 1, X86InstInfo{"FEMMS", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
@@ -44,7 +44,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
{0x16, 1, X86InstInfo{"MOVLHPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
{0x17, 1, X86InstInfo{"MOVHPS", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
{0x18, 1, X86InstInfo{"", TYPE_GROUP_16, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x19, 7, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x19, 7, X86InstInfo{"NOP", TYPE_INST, FLAGS_DEBUG | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x20, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x22, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
@@ -59,7 +59,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
{0x2F, 1, X86InstInfo{"COMISS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
{0x30, 1, X86InstInfo{"WRMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x31, 1, X86InstInfo{"RDTSC", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x31, 1, X86InstInfo{"RDTSC", TYPE_INST, FLAGS_DEBUG | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x32, 1, X86InstInfo{"RDMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x33, 1, X86InstInfo{"RDPMC", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x34, 1, X86InstInfo{"SYSENTER", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
@@ -166,7 +166,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
{0x9E, 1, X86InstInfo{"SETLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x9F, 1, X86InstInfo{"SETNLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_DEBUG | FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA4, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1, nullptr}},
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0, nullptr}},
@@ -254,7 +254,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
{0xFC, 1, X86InstInfo{"PADDB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xFD, 1, X86InstInfo{"PADDW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xFE, 1, X86InstInfo{"PADDD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
// FEX reserved instructions
// Unused x86 encoding instruction.
@@ -279,15 +279,9 @@ namespace InstFlags {
using InstFlagType = uint64_t;
constexpr InstFlagType FLAGS_NONE = 0;
// The secondary Opcode Map uses prefix bytes to overlay more instruction
// But some instructions need to ignore this overlay and consume these prefixes.
constexpr InstFlagType FLAGS_NO_OVERLAY = (1ULL << 0);
// Some instructions partially ignore overlay
// Ignore OpSize (0x66) in this case
constexpr InstFlagType FLAGS_NO_OVERLAY66 = (1ULL << 1);
constexpr InstFlagType FLAGS_DEBUG = (1ULL << 1);
constexpr InstFlagType FLAGS_DEBUG_MEM_ACCESS = (1ULL << 2);
// Only SEXT if the instruction is operating in 64bit operand size
constexpr InstFlagType FLAGS_SRC_SEXT64BIT = (1ULL << 3);
constexpr InstFlagType FLAGS_SUPPORTS_REP = (1ULL << 3);
constexpr InstFlagType FLAGS_BLOCK_END = (1ULL << 4);
constexpr InstFlagType FLAGS_SETS_RIP = (1ULL << 5);
@@ -337,17 +331,27 @@ constexpr InstFlagType FLAGS_MODRM = (1ULL << 16);
constexpr InstFlagType FLAGS_SF_MOD_MEM_ONLY = (1ULL << 18);
constexpr InstFlagType FLAGS_SF_MOD_REG_ONLY = (1ULL << 19);
// The secondary Opcode Map uses prefix bytes to overlay more instruction
// But some instructions need to ignore this overlay and consume these prefixes.
constexpr InstFlagType FLAGS_NO_OVERLAY = (1ULL << 20);
// Some instructions partially ignore overlay
// Ignore OpSize (0x66) in this case
constexpr InstFlagType FLAGS_NO_OVERLAY66 = (1ULL << 21);
// x87
constexpr InstFlagType FLAGS_POP = (1ULL << 20);
constexpr InstFlagType FLAGS_POP = (1ULL << 22);
// Only SEXT if the instruction is operating in 64bit operand size
constexpr InstFlagType FLAGS_SRC_SEXT64BIT = (1ULL << 23);
// Whether or not the instruction has a VEX prefix for the first source operand
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 21);
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 24);
// Whether or not the instruction has a VEX prefix for the second source operand
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 22);
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 25);
// Whether or not the instruction has a VEX prefix for the destination
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 23);
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 26);
// Whether or not the instruction has a VSIB byte
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 24);
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 27);
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
@@ -415,12 +419,35 @@ constexpr uint8_t OpToIndex(uint8_t Op) {
using DecodedOp = DecodedInst const*;
using OpDispatchPtr = void (IR::OpDispatchBuilder::*)(DecodedOp);
#ifndef NDEBUG
namespace X86InstDebugInfo {
constexpr uint64_t FLAGS_MEM_ALIGN_4 = (1 << 0);
constexpr uint64_t FLAGS_MEM_ALIGN_8 = (1 << 1);
constexpr uint64_t FLAGS_MEM_ALIGN_16 = (1 << 2);
constexpr uint64_t FLAGS_MEM_ALIGN_SIZE = (1 << 3); // If instruction size changes depending on prefixes
constexpr uint64_t FLAGS_MEM_ACCESS = (1 << 4);
constexpr uint64_t FLAGS_DEBUG = (1 << 5);
constexpr uint64_t FLAGS_DIVIDE = (1 << 6);
struct Flags {
uint64_t DebugFlags;
};
void InstallDebugInfo();
}
#endif
struct X86InstInfo {
char const *Name;
InstType Type;
InstFlags::InstFlagType Flags; ///< Must be larger than InstFlags enum
uint8_t MoreBytes;
OpDispatchPtr OpcodeDispatcher;
#ifndef NDEBUG
X86InstDebugInfo::Flags DebugInfo;
uint32_t NumUnitTestsGenerated;
#endif
bool operator==(const X86InstInfo &b) const {
if (strcmp(Name, b.Name) != 0 ||
@@ -497,6 +524,12 @@ extern std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps;
// EVEX
extern std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps;
#ifndef NDEBUG
extern uint64_t Total;
extern uint64_t NumInsts;
#endif
template <typename OpcodeType>
struct X86TablesInfoStruct {
OpcodeType first;
@@ -515,6 +548,11 @@ static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInfoStruct<Op
for (uint32_t i = 0; i < Op.second; ++i) {
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
FinalTable[OpNum + i] = Info;
#ifndef NDEBUG
++Total;
if (Info.Type == TYPE_INST)
NumInsts++;
#endif
}
}
};
@@ -532,6 +570,11 @@ static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86TablesInfoS
}
else {
FinalTable[OpNum + i] = Info;
#ifndef NDEBUG
++Total;
if (Info.Type == TYPE_INST)
NumInsts++;
#endif
}
}
}
@@ -559,6 +602,11 @@ static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct
}
}
}
#ifndef NDEBUG
++Total;
if (Info.Type == TYPE_INST)
NumInsts++;
#endif
}
}
};
+12 -3
View File
@@ -363,9 +363,18 @@ namespace FEXCore::IR {
// Insert to caches if we generated IR
if (GeneratedIR) {
// If the IR doesn't need to be retained then we can just delete it now
delete DebugData;
if (IRList->IsCopy()) delete IRList;
if (CTX->GetGdbServerStatus()) {
// Add to thread local ir cache
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), std::move(RAData), decltype(Entry.DebugData)(DebugData)};
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
Thread->DebugStore.insert({GuestRIP, std::move(Entry)});
}
else {
// If the IR doesn't need to be retained then we can just delete it now
delete DebugData;
if (IRList->IsCopy()) delete IRList;
}
}
}
+54 -96
View File
@@ -67,6 +67,8 @@
"constexpr uint8_t COND_SLT = 11",
"constexpr uint8_t COND_SGT = 12",
"constexpr uint8_t COND_SLE = 13",
"constexpr uint8_t COND_ANDZ = 14 /* (a & b) == 0 */",
"constexpr uint8_t COND_ANDNZ = 15 /* (a & b) != 0 */",
"constexpr uint8_t COND_FLU = 16 /* float less or unordred */",
"constexpr uint8_t COND_FGE = 17 /* float greater or equal */",
@@ -75,8 +77,6 @@
"constexpr uint8_t COND_FU = 20 /* float unordred */",
"constexpr uint8_t COND_FNU = 21 /* float not unordred */",
"constexpr uint8_t COND_AL = 32 /* always */",
"constexpr FEXCore::IR::RegisterClassType GPRClass {0}",
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {1}",
"constexpr FEXCore::IR::RegisterClassType FPRClass {2}",
@@ -264,7 +264,7 @@
"HasSideEffects": true,
"RAOverride": "0"
},
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, u8:$CompareSize{0}, i1:$FromNZCV{false}": {
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, u8:$CompareSize{0}": {
"HasSideEffects": true,
"RAOverride": "2",
"EmitValidation": [
@@ -447,17 +447,6 @@
]
},
"GPR = LoadNZCV": {
"Desc": ["Loads value of NZCV register"],
"DestSize": "4"
},
"StoreNZCV GPR:$Value": {
"HasSideEffects": true,
"Desc": ["Stores value to NZCV register"],
"DestSize": "4"
},
"GPR = LoadFlag u32:$Flag": {
"Desc": ["Loads an x86-64 flag from the context object",
"Specialized to allow flexible implementation of flag handling"
@@ -865,9 +854,9 @@
"DestSize": "8"
},
"GPR = Neg OpSize:#Size, GPR:$Src, CondClass:$Cond{{COND_AL}}": {
"Desc": ["Integer negation, with optional predication",
"Dest = Cond ? -Src : Src",
"GPR = Neg OpSize:#Size, GPR:$Src": {
"Desc": ["Integer negation",
"Dest = -Src",
"Will truncate to 64 or 32bits"
],
"DestSize": "Size",
@@ -875,6 +864,17 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Abs OpSize:#Size, GPR:$Src": {
"Desc": ["Integer 2's complement absolute value",
"Dest = std::abs(Src)",
"Will truncate to 64 or 32bits"
],
"DestSize": "Size",
"ImplicitFlagClobber": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Not OpSize:#Size, GPR:$Src": {
"Desc": ["Integer binary not",
"op:",
@@ -953,48 +953,28 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"AddNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the sum of two GPRs"],
"HasSideEffects": true,
"DestSize": "Size",
"GPR = AddNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Return NZCV for the sum of two GPRs"],
"DestSize": "4",
"ImplicitFlagClobber": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
]
},
"CarryInvert": {
"Desc": ["Invert carry flag in NZCV"],
"HasSideEffects": true
},
"AXFlag": {
"Desc": ["After an FCmp, converts NZCV flags from the Arm format to a mysterious eXternal format"],
"HasSideEffects": true
},
"RmifNZCV GPR:$Src, u8:$Rotate, u8:$Mask": {
"Desc": ["Rotate, mask, and insert into NZCV on FlagM platforms"],
"HasSideEffects": true
},
"CondAddNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2, CondClass:$Cond, u8:$FalseNZCV": {
"Desc": ["If condition is true, set NZCV per sum of GPRs, else force NZCV to a constant."],
"HasSideEffects": true,
"DestSize": "Size",
"GPR = AdcNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
"DestSize": "4",
"ImplicitFlagClobber": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
]
},
"AdcNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the sum of two GPRs and carry-in given as NZCV"],
"HasSideEffects": true,
"DestSize": "Size",
"GPR = SbbNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
"DestSize": "4",
"ImplicitFlagClobber": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"SbbNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the difference of two GPRs and carry-in given as NZCV"],
"HasSideEffects": true,
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Sub OpSize:#Size, GPR:$Src1, GPR:$Src2": {
@@ -1016,14 +996,14 @@
"_Shift != ShiftType::ROR"
]
},
"SubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the difference of two GPRs. ",
"Carry flag uses arm64 definition, inverted x86.",
"GPR = SubNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, u8:$InvertCarry": {
"Desc": ["Return NZCV for the difference of two GPRs. ",
"If InvertCarry is nonzero, carry flag uses x86 definition, inverted from arm64.",
""],
"DestSize": "Size",
"HasSideEffects": true,
"DestSize": "4",
"ImplicitFlagClobber": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Or OpSize:#Size, GPR:$Src1, GPR:$Src2": {
@@ -1066,13 +1046,6 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = XorShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
"Desc": [ "Integer binary exclusive or with shifted register"],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = And OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer binary and"
],
@@ -1088,10 +1061,10 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"TestNZ OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the binary AND of two GPRs, setting N and Z accordingly and zeroing C and V"],
"DestSize": "Size",
"HasSideEffects": true
"GPR = TestNZ u8:$Size, GPR:$Src1": {
"Desc": ["Return NZCV for a GPR, setting N and Z accordingly and zeroing C and V"],
"ImplicitFlagClobber": true,
"DestSize": "4"
},
"GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer logical shift left"
@@ -1236,16 +1209,6 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = NZCVSelect OpSize:#ResultSize, CondClass:$Cond, GPR:$TrueVal, GPR:$FalseVal": {
"Desc": ["Select based on value in NZCV flags",
"op:",
"Dest = Cond ? TrueVal : FalseVal"
],
"DestSize": "ResultSize",
"EmitValidation": [
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Select OpSize:#ResultSize, OpSize:$CompareSize, CondClass:$Cond, SSA:$Cmp1, SSA:$Cmp2, GPR:$TrueVal, GPR:$FalseVal": {
"Desc": ["Ternary selection of GPRs",
"op:",
@@ -1357,12 +1320,12 @@
"DestSize": "DestElementSize"
},
"FCmp u8:$ElementSize, FPR:$Scalar1, FPR:$Scalar2": {
"Desc": ["Does a scalar unordered compare and sets NZCV accordingly.",
"NZCV follows Arm conventions, a separate AXFLAG instruction is required for x86",
"GPR = FCmp u8:$ElementSize, FPR:$Scalar1, FPR:$Scalar2, u32:$Flags": {
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
"Ordering flag result is true if either float input is NaN"
],
"HasSideEffects": true
"ImplicitFlagClobber": true,
"DestSize": "4"
}
},
"VectorScalar": {
@@ -1618,6 +1581,7 @@
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VCMPEQZ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -1626,6 +1590,7 @@
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
"Compares the vector against zero"
],
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -1634,6 +1599,7 @@
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
"Compares the vector against zero"
],
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -1650,10 +1616,6 @@
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VUShraI u8:#RegisterSize, u8:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VSShrI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
@@ -1886,10 +1848,12 @@
},
"FPR = VFMin u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFMax u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -2000,7 +1964,8 @@
"FPR = VCMPEQ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
"NumElements": "RegisterSize / ElementSize",
"ImplicitFlagClobber": true
},
"FPR = VCMPGT u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
@@ -2008,6 +1973,7 @@
"Each element is compared, if the result is true then the resulting element is ~0, else zero"
],
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -2187,14 +2153,6 @@
"Desc": "Assists in key generation",
"DestSize": "16"
},
"FPR = VSha1H FPR:$Src": {
"Desc": "Does vector scalar SHA1H instruction",
"DestSize": "FEXCore::IR::OpSize::i32Bit"
},
"FPR = VSha256U0 FPR:$Src1, FPR:$Src2": {
"Desc": "Does vector scalar VSha256U0 instruction",
"DestSize": "FEXCore::IR::OpSize::i128Bit"
},
"GPR = CRC32 GPR:$Src1, GPR:$Src2, u8:$SrcSize": {
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
],
-5
View File
@@ -44,11 +44,6 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
}
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, CondClassType Arg) {
if (Arg == COND_AL) {
*out << "ALWAYS";
return;
}
static constexpr std::array<std::string_view, 22> CondNames = {
"EQ",
"NEQ",
+125 -101
View File
@@ -194,6 +194,8 @@ public:
private:
bool HandleConstantPools(IREmitter *IREmit, const IRListView& CurrentIR);
void CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR);
void FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR);
void LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR);
bool ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& CurrentIR,
OrderedNode* CodeNode, IROp_Header* IROp);
@@ -265,6 +267,92 @@ bool ConstProp::HandleConstantPools(IREmitter *IREmit, const IRListView& Current
return Changed;
}
// Code motion around selects
// Moves unary ops that depend on a select before the select, if both inputs are constants
// assumes that unary ops without side effects on constants will be constprop'd
void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR) {
// Code motion around selects
// Moves unary ops that depend on a select before the select, if both inputs are constants
// assumes that unary ops without side effects on constants will be constprop'd
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
auto BlockOp = BlockIROp->CW<FEXCore::IR::IROp_CodeBlock>();
for (auto [UnaryOpNode, UnaryOpHdr] : CurrentIR.GetCode(BlockNode)) {
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)
&& !ImplicitFlagClobber(UnaryOpHdr->Op)) {
// could be moved
auto SelectOpNode = IREmit->UnwrapNode(UnaryOpHdr->Args[0]);
auto SelectOpHdr = IREmit->GetOpHeader(UnaryOpHdr->Args[0]);
auto SelectOp = SelectOpHdr->CW<IR::IROp_Select>();
// the value isn't used after the select otherwise
// make sure the sizes match
if (SelectOpHdr->Size == UnaryOpHdr->Size && SelectOpHdr->Op == OP_SELECT && SelectOpNode->NumUses == 1
&& IREmit->IsValueConstant(SelectOp->TrueVal)
&& IREmit->IsValueConstant(SelectOp->FalseVal)) {
IREmit->SetWriteCursor(IREmit->UnwrapNode(SelectOpNode->Header.Previous));
size_t OpSize = FEXCore::IR::GetSize(UnaryOpHdr->Op);
/// copy for TrueVal ///
auto NewUnaryOp1 = IREmit->AllocateRawOp(OpSize);
// Copy over the op
memcpy(NewUnaryOp1.first, UnaryOpHdr, OpSize);
for (int i = 0; i < IR::GetArgs(NewUnaryOp1.first->Op); i++) {
NewUnaryOp1.first->Args[i] = IREmit->WrapNode(IREmit->Invalid());
}
// Set New Op to operate on the constant
IREmit->ReplaceNodeArgument(NewUnaryOp1, 0, IREmit->UnwrapNode(SelectOp->TrueVal));
// Make select use the operated constant
IREmit->ReplaceNodeArgument(SelectOpNode, 2, NewUnaryOp1);
/// copy for FalseVal ///
auto NewUnaryOp2 = IREmit->AllocateRawOp(OpSize);
// Copy over the op
memcpy(NewUnaryOp2.first, UnaryOpHdr, OpSize);
for (int i = 0; i < IR::GetArgs(NewUnaryOp2.first->Op); i++) {
NewUnaryOp2.first->Args[i] = IREmit->WrapNode(IREmit->Invalid());
}
// Set New Op to operate on the constant
IREmit->ReplaceNodeArgument(NewUnaryOp2, 0, IREmit->UnwrapNode(SelectOp->FalseVal));
// Make select use the operated constant
IREmit->ReplaceNodeArgument(SelectOpNode, 3, NewUnaryOp2);
// Replace uses of the defuct unary op w/ select
IREmit->ReplaceAllUsesWithRange(UnaryOpNode, SelectOpNode, IREmit->GetIterator(IREmit->WrapNode(UnaryOpNode)), IREmit->GetIterator(BlockOp->Last));
}
}
}
}
}
void ConstProp::FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR) {
// Make all FCMPs set no flags
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
if (IROp->Op == OP_FCMP) {
auto fcmp = IROp->CW<IR::IROp_FCmp>();
fcmp->Flags = 0;
}
}
// Set needed flags
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
if (IROp->Op == OP_GETHOSTFLAG) {
auto ghf = IROp->CW<IR::IROp_GetHostFlag>();
auto fcmp = IREmit->GetOpHeader(ghf->Value)->CW<IR::IROp_FCmp>();
LOGMAN_THROW_AA_FMT(fcmp->Header.Op == OP_FCMP || fcmp->Header.Op == OP_F80CMP, "Unexpected OP_GETHOSTFLAG source");
if(fcmp->Header.Op == OP_FCMP) {
fcmp->Flags |= 1 << ghf->Flag;
}
}
}
}
// LoadMem / StoreMem imm pooling
// If imms are close by, use address gen to generate the values instead of using a new imm
void ConstProp::LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR) {
@@ -645,8 +733,6 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
}
break;
}
/* TODO: restore this when we have rmif or something? */
#if 0
case OP_TESTNZ: {
auto Op = IROp->CW<IR::IROp_TestNZ>();
uint64_t Constant1{};
@@ -661,7 +747,6 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
}
break;
}
#endif
case OP_OR: {
auto Op = IROp->CW<IR::IROp_Or>();
uint64_t Constant1{};
@@ -920,6 +1005,37 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
}
break;
}
case OP_CONDJUMP: {
auto Op = IROp->CW<IR::IROp_CondJump>();
auto Select = IREmit->GetOpHeader(Op->Header.Args[0]);
uint64_t Constant;
// Fold the select into the CondJump if possible. Could handle more complex cases, too.
if (Op->Cond.Val == COND_NEQ && IREmit->IsValueConstant(Op->Cmp2, &Constant) && Constant == 0 && Select->Op == OP_SELECT) {
const auto SelectCmpClass = IREmit->WalkFindRegClass(Select->Args[0]);
if (SelectCmpClass == GPRPairClass) {
// If the comparison class is a GPRPair then don't fold the select since it isn't free.
break;
}
uint64_t Constant1{};
uint64_t Constant2{};
if (IREmit->IsValueConstant(Select->Args[2], &Constant1) && IREmit->IsValueConstant(Select->Args[3], &Constant2)) {
if (Constant1 == 1 && Constant2 == 0) {
auto slc = Select->C<IR::IROp_Select>();
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(Select->Args[0]));
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->UnwrapNode(Select->Args[1]));
Op->Cond = slc->Cond;
Op->CompareSize = slc->CompareSize;
Changed = true;
}
}
}
break;
}
default:
break;
}
@@ -986,54 +1102,16 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
break;
}
case OP_CONDADDNZCV:
{
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
uint64_t Constant2{};
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
if (IsImmAddSub(Constant2)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
Changed = true;
}
}
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
if (Constant1 == 0) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
Changed = true;
}
}
break;
}
case OP_TESTNZ:
{
auto Op = IROp->C<IR::IROp_TestNZ>();
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
if (IsImmLogical(Constant1, IROp->Size * 8)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
Changed = true;
}
}
break;
}
case OP_SELECT:
{
auto Op = IROp->C<IR::IROp_Select>();
bool Bitwise = Op->Cond == COND_ANDZ ||
Op->Cond == COND_ANDNZ;
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
if (IsImmAddSub(Constant1)) {
if (Bitwise ? IsImmLogical(Constant1, IROp->Size * 8) : IsImmAddSub(Constant1)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
@@ -1064,33 +1142,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
break;
}
case OP_NZCVSELECT:
{
auto Op = IROp->C<IR::IROp_NZCVSelect>();
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
// We always allow source 1 to be zero, but source 0 can only be a
// special 1/~0 constant if source 1 is 0.
uint64_t Constant0{};
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1) &&
Constant1 == 0)
{
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant0) &&
(Constant0 == 1 || Constant0 == AllOnes))
{
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, Constant0));
}
}
break;
}
case OP_CONDJUMP:
{
auto Op = IROp->C<IR::IROp_CondJump>();
@@ -1218,35 +1269,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
}
break;
}
case OP_MEMCPY:
{
auto Op = IROp->CW<IR::IROp_MemCpy>();
uint64_t Constant{};
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
Changed = true;
}
break;
}
case OP_MEMSET:
{
auto Op = IROp->CW<IR::IROp_MemSet>();
uint64_t Constant{};
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
Changed = true;
}
break;
}
default:
break;
}
@@ -1266,6 +1288,8 @@ bool ConstProp::Run(IREmitter *IREmit) {
Changed = true;
}
CodeMotionAroundSelects(IREmit, CurrentIR);
FCMPOptimization(IREmit, CurrentIR);
LoadMemStoreMemImmediatePooling(IREmit, CurrentIR);
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
@@ -277,24 +277,6 @@ namespace {
});
}
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, pf_raw),
sizeof(FEXCore::Core::CPUState::pf_raw),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, af_raw),
sizeof(FEXCore::Core::CPUState::af_raw),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
@@ -437,10 +419,6 @@ namespace {
SetAccess(Offset++, LastAccessType::NONE);
}
// PF/AF
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
SetAccess(Offset++, LastAccessType::NONE);
}
@@ -6,7 +6,6 @@ $end_info$
*/
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "FEXCore/Core/X86Enums.h"
#include "Interface/IR/Passes.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
@@ -519,21 +518,12 @@ namespace {
const auto GetRegAndClassFromOffset = [&, this](uint32_t Offset) {
const auto beginGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0]);
const auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[16]);
const auto pf = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
const auto af = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
const auto [beginFpr, endFpr] = GetFPRBeginAndEnd();
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr) || (Offset == pf) || (Offset == af), "Unexpected Offset {}", Offset);
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr), "Unexpected Offset {}", Offset);
unsigned FlagOffset =
Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount - 2;
if (Offset == pf) {
return PhysicalRegister(GPRFixedClass, FlagOffset);
} else if (Offset == af) {
return PhysicalRegister(GPRFixedClass, FlagOffset + 1);
} else if (Offset >= beginGpr && Offset < endGpr) {
if (Offset >= beginGpr && Offset < endGpr) {
auto reg = (Offset - beginGpr) / Core::CPUState::GPR_REG_SIZE;
return PhysicalRegister(GPRFixedClass, reg);
} else if (Offset >= beginFpr && Offset < endFpr) {
@@ -554,21 +544,12 @@ namespace {
const auto GetStaticMapFromOffset = [&](uint32_t Offset) -> LiveRange** {
const auto beginGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0]);
const auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[16]);
const auto pf = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
const auto af = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
const auto [beginFpr, endFpr] = GetFPRBeginAndEnd();
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr) || (Offset == pf) || (Offset == af), "Unexpected Offset {}", Offset);
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr), "Unexpected Offset {}", Offset);
unsigned FlagOffset =
Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount - 2;
if (Offset == pf) {
return &StaticMaps[FlagOffset];
} else if (Offset == af) {
return &StaticMaps[FlagOffset + 1];
} else if (Offset >= beginGpr && Offset < endGpr) {
if (Offset >= beginGpr && Offset < endGpr) {
auto reg = (Offset - beginGpr) / Core::CPUState::GPR_REG_SIZE;
return &StaticMaps[reg];
} else if (Offset >= beginFpr && Offset < endFpr) {
@@ -5,8 +5,8 @@
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/Utils/DeferredSignalMutex.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <FEXCore/fextl/memory.h>
@@ -272,7 +272,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
size_t NumberOfPages = length / FHU::FEX_PAGE_SIZE;
// This needs a mutex to be thread safe
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
uint64_t AllocatedOffset{};
LiveVMARegion *LiveRegion{};
@@ -460,7 +460,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
}
// This needs a mutex to be thread safe
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE);
@@ -585,7 +585,7 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
OSAllocator_64Bit::~OSAllocator_64Bit() {
// This needs a mutex to be thread safe
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
// Walk the pages and deallocate
// First walk the live regions
@@ -155,6 +155,13 @@ namespace CPU {
*/
[[nodiscard]] virtual void *MapRegion(void *HostPtr, uint64_t GuestPtr, uint64_t Size) = 0;
/**
* @brief This is post-setup initialization that is called just before code executino
*
* Guest memory is available at this point and ThreadState is valid
*/
virtual void Initialize() {}
/**
* @brief Lets FEXCore know if this CPUBackend needs IR and DebugData for CompileCode
*
+48 -27
View File
@@ -116,6 +116,16 @@ namespace FEXCore::Context {
*/
FEX_DEFAULT_VISIBILITY static fextl::unique_ptr<FEXCore::Context::Context> CreateNewContext();
/**
* @brief Post creation context initialization
* Once configurations have been set, do the post-creation initialization with that configuration
*
* @param CTX The context that we created
*
* @return true if we managed to initialize correctly
*/
FEX_DEFAULT_VISIBILITY virtual bool InitializeContext() = 0;
/**
* @brief Allows setting up in memory code and other things prior to launchign code execution
*
@@ -191,6 +201,25 @@ namespace FEXCore::Context {
FEX_DEFAULT_VISIBILITY virtual void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) = 0;
FEX_DEFAULT_VISIBILITY virtual void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) = 0;
/**
* @brief Gets the program exit status
*
*
* @param CTX The context that we created
*
* @return The program exit status
*/
FEX_DEFAULT_VISIBILITY virtual int GetProgramStatus() const = 0;
/**
* @brief [[threadsafe]] Returns the ExitReason of the parent thread. Typically used for async result status
*
* @param CTX The context that we created
*
* @return The ExitReason for the parentthread
*/
FEX_DEFAULT_VISIBILITY virtual ExitReason GetExitReason() = 0;
/**
* @brief [[theadsafe]] Checks if the Context is either done working or paused(in the case of single stepping)
*
@@ -202,6 +231,22 @@ namespace FEXCore::Context {
*/
FEX_DEFAULT_VISIBILITY virtual bool IsDone() const = 0;
/**
* @brief Gets a copy the CPUState of the parent thread
*
* @param CTX The context that we created
* @param State The state object to populate
*/
FEX_DEFAULT_VISIBILITY virtual void GetCPUState(FEXCore::Core::CPUState *State) const = 0;
/**
* @brief Copies the CPUState provided to the parent thread
*
* @param CTX The context that we created
* @param State The satate object to copy from
*/
FEX_DEFAULT_VISIBILITY virtual void SetCPUState(const FEXCore::Core::CPUState *State) = 0;
/**
* @brief Allows the frontend to pass in a custom CPUBackend creation factory
*
@@ -225,36 +270,12 @@ namespace FEXCore::Context {
///< State reconstruction helpers
///< Reconstructs the guest RIP from the passed in thread context and related Host PC.
FEX_DEFAULT_VISIBILITY virtual uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) = 0;
/**
* @brief Reconstructs a compacted EFLAGS from FEX's internal EFLAG representation.
*
* @param Thread The thread getting the state reconstructed
* @param WasInJIT If the code was in the JIT at the time.
* @param HostGPRs The host Arm64 GPRs at the point of state inside the JIT.
* @param PSTATE The Arm64 PState value.
*
* If WasInJIT is false then HostGPRs and PSTATE is ignored, with the assumption that the FEX JIT has already stored all state in to the
* ThreadState object.
*
* @return x86 EFLAGS reconstructed
*/
FEX_DEFAULT_VISIBILITY virtual uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) = 0;
///< Reconstructs a compacted EFLAGS from FEX's internal EFLAG representation.
FEX_DEFAULT_VISIBILITY virtual uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) = 0;
///< Sets FEX's internal EFLAGS representation to the passed in compacted form.
FEX_DEFAULT_VISIBILITY virtual void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) = 0;
/**
* @brief Create a new thread object that doesn't inherit any state.
* Used to create FEX thread objects in preparation for creating a true OS thread.
*
* @param InitialRIP The starting RIP of this thread
* @param StackPointer The starting RSP of this thread
* @param NewThreadState The thread state to inherit from if not nullptr.
* @param ParentTID The thread ID that the parent is inheriting from
*
* @return A new InternalThreadState object for using with a new guest thread.
*/
FEX_DEFAULT_VISIBILITY virtual FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState = nullptr, uint64_t ParentTID = 0) = 0;
FEX_DEFAULT_VISIBILITY virtual FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) = 0;
FEX_DEFAULT_VISIBILITY virtual void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) = 0;
FEX_DEFAULT_VISIBILITY virtual void InitializeThread(FEXCore::Core::InternalThreadState *Thread) = 0;
FEX_DEFAULT_VISIBILITY virtual void RunThread(FEXCore::Core::InternalThreadState *Thread) = 0;
+4 -3
View File
@@ -72,7 +72,7 @@ namespace FEXCore::Core {
static_assert(std::is_trivially_copyable_v<NonAtomicRefCounter<uint64_t>>, "needs to be trivially copyable");
static_assert(sizeof(NonAtomicRefCounter<uint64_t>) == sizeof(uint64_t), "Needs to be correct size");
struct CPUState {
struct FEX_PACKED CPUState {
// Allows more efficient handling of the register
// file in the event AVX is not supported.
union XMMRegs {
@@ -102,8 +102,6 @@ namespace FEXCore::Core {
uint64_t InlineJITBlockHeader{};
XMMRegs xmm{};
uint8_t flags[48]{};
uint64_t pf_raw{};
uint64_t af_raw{};
uint64_t mm[8][2]{};
// 32bit x86 state
@@ -337,4 +335,7 @@ namespace FEXCore::Core {
static_assert(sizeof(CpuStateFrame::SynchronousFaultData) == 8, "This needs to be 8 bytes");
static_assert(std::alignment_of_v<CpuStateFrame::SynchronousFaultDataStruct> == 8, "This needs to be 8 bytes");
static_assert(offsetof(CpuStateFrame, SynchronousFaultData) % 8 == 0, "This needs to be aligned");
FEX_DEFAULT_VISIBILITY std::string_view const& GetFlagName(unsigned Flag);
FEX_DEFAULT_VISIBILITY std::string_view const& GetGRegName(unsigned Reg);
}
@@ -35,6 +35,8 @@ namespace Core {
#endif
};
}
using HostSignalDelegatorFunction = std::function<bool(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext)>;
class SignalDelegator {
public:
virtual ~SignalDelegator() = default;
@@ -47,6 +49,16 @@ namespace Core {
virtual void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
virtual void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
/**
* @brief Registers a signal handler for the host to handle a signal
*
* It's a process level signal handler so one must be careful
*/
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
// Called from the thunk handler to handle the signal
void HandleSignal(int Signal, void *Info, void *UContext);
/**
* @brief Check to ensure the XID handler is still set to the FEX handler
*
@@ -55,6 +67,12 @@ namespace Core {
*/
virtual void CheckXIDHandler() = 0;
constexpr static size_t MAX_SIGNALS {64};
// Use the last signal just so we are less likely to ever conflict with something that the guest application is using
// 64 is used internally by Valgrind
constexpr static size_t SIGNAL_FOR_PAUSE {63};
struct SignalDelegatorConfig {
bool StaticRegisterAllocation{};
bool SupportsAVX{};
@@ -106,5 +124,31 @@ namespace Core {
protected:
SignalDelegatorConfig Config;
virtual FEXCore::Core::InternalThreadState *GetTLSThread() = 0;
virtual void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) = 0;
/**
* @brief Registers a signal handler for the host to handle a signal
*
* It's a process level signal handler so one must be careful
*/
virtual void FrontendRegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) = 0;
virtual void FrontendRegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) = 0;
private:
struct HostSignalHandler {
fextl::vector<FEXCore::HostSignalDelegatorFunction> Handlers{};
FEXCore::HostSignalDelegatorFunction FrontendHandler{};
};
std::array<HostSignalHandler, MAX_SIGNALS + 1> HostHandlers{};
protected:
void SetHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
HostHandlers[Signal].Handlers.push_back(std::move(Func));
}
void SetFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
HostHandlers[Signal].FrontendHandler = std::move(Func);
}
};
}
+1 -1
View File
@@ -334,7 +334,7 @@ friend class FEXCore::IR::PassManager;
return Ptr;
}
virtual void SaveNZCV(IROps Op) {
virtual void SaveNZCV() {
// Overriden by dispatcher, stubbed for IR tests
}
@@ -0,0 +1,338 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Debug/InternalThreadState.h>
#include <atomic>
#include <cstdint>
#include <mutex>
#include <shared_mutex>
#include <signal.h>
#ifndef _WIN32
#include <sys/syscall.h>
#endif
#include <unistd.h>
namespace FEXCore {
#ifndef _WIN32
// Replacement for std::mutexes to deal with unlocking issues in the face of Linux fork() semantics.
//
// A fork() only clones the parent's calling thread. Other threads are silently dropped, which permanently leaves any mutexes owned by them locked.
// To address this issue, ForkableUniqueMutex and ForkableSharedMutex provide a way to forcefully remove any dangling locks and reset the mutexes to their default state.
class ForkableUniqueMutex final {
public:
ForkableUniqueMutex()
: Mutex (PTHREAD_MUTEX_INITIALIZER) {
}
// Move-only type
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = default;
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = default;
void lock() {
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
}
void unlock() {
[[maybe_unused]] const auto Result = pthread_mutex_unlock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
}
// Initialize the internal pthread object to its default initializer state.
// Should only ever be used in the child process when a Linux fork() has occured.
void StealAndDropActiveLocks() {
Mutex = PTHREAD_MUTEX_INITIALIZER;
}
private:
pthread_mutex_t Mutex;
};
class ForkableSharedMutex final {
public:
ForkableSharedMutex()
: Mutex (PTHREAD_RWLOCK_INITIALIZER) {
}
// Move-only type
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
ForkableSharedMutex(ForkableSharedMutex &&rhs) = default;
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = default;
void lock() {
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
}
void unlock() {
[[maybe_unused]] const auto Result = pthread_rwlock_unlock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
}
void lock_shared() {
[[maybe_unused]] const auto Result = pthread_rwlock_rdlock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
}
void unlock_shared() {
unlock();
}
bool try_lock() {
const auto Result = pthread_rwlock_trywrlock(&Mutex);
return Result == 0;
}
bool try_lock_shared() {
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
return Result == 0;
}
// Initialize the internal pthread object to its default initializer state.
// Should only ever be used in the child process when a Linux fork() has occured.
void StealAndDropActiveLocks() {
Mutex = PTHREAD_RWLOCK_INITIALIZER;
}
private:
pthread_rwlock_t Mutex;
};
#else
// Windows doesn't support forking, so these can be standard mutexes.
class ForkableUniqueMutex final {
public:
ForkableUniqueMutex() = default;
// Non-moveable
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = delete;
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = delete;
void lock() {
Mutex.lock();
}
void unlock() {
Mutex.unlock();
}
// Initialize the internal pthread object to its default initializer state.
// Should only ever be used in the child process when a Linux fork() has occured.
void StealAndDropActiveLocks() {
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
}
private:
std::mutex Mutex;
};
class ForkableSharedMutex final {
public:
ForkableSharedMutex() = default;
// Non-moveable
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
ForkableSharedMutex(ForkableSharedMutex &&rhs) = delete;
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = delete;
void lock() {
Mutex.lock();
}
void unlock() {
Mutex.unlock();
}
void lock_shared() {
Mutex.lock_shared();
}
void unlock_shared() {
Mutex.unlock_shared();
}
bool try_lock() {
return Mutex.try_lock();
}
bool try_lock_shared() {
return Mutex.try_lock_shared();
}
// Initialize the internal pthread object to its default initializer state.
// Should only ever be used in the child process when a Linux fork() has occured.
void StealAndDropActiveLocks() {
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
}
private:
std::shared_mutex Mutex;
};
#endif
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
class ScopedDeferredSignalWithMutexBase final {
public:
ScopedDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread)
: Mutex {&_Mutex}
, Thread {Thread} {
// Needs to be atomic so that operations can't end up getting reordered around this.
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
// Lock the mutex
(Mutex->*lock_fn)();
}
// No copy or assignment possible
ScopedDeferredSignalWithMutexBase(const ScopedDeferredSignalWithMutexBase&) = delete;
ScopedDeferredSignalWithMutexBase& operator=(ScopedDeferredSignalWithMutexBase&) = delete;
// Only move
ScopedDeferredSignalWithMutexBase(ScopedDeferredSignalWithMutexBase &&rhs)
: Mutex {rhs.Mutex}
, Thread {rhs.Thread} {
rhs.Mutex = nullptr;
}
~ScopedDeferredSignalWithMutexBase() {
if (Mutex != nullptr) {
// Unlock the mutex
(Mutex->*unlock_fn)();
#ifdef _M_X86_64
// Needs to be atomic so that operations can't end up getting reordered around this.
// Without this, the recount and the signal access could get reordered.
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
// X86-64 must do an additional check around the store.
if ((Result - 1) == 0) {
// Must happen after the refcount store
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
}
#else
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
#endif
}
}
private:
MutexType *Mutex;
FEXCore::Core::InternalThreadState *Thread;
};
using ScopedDeferredSignalWithMutex = ScopedDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
using ScopedDeferredSignalWithSharedLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
using ScopedDeferredSignalWithUniqueLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
// Forkable variant
using ScopedDeferredSignalWithForkableMutex = ScopedDeferredSignalWithMutexBase<
FEXCore::ForkableUniqueMutex,
&FEXCore::ForkableUniqueMutex::lock,
&FEXCore::ForkableUniqueMutex::unlock>;
using ScopedDeferredSignalWithForkableSharedLock = ScopedDeferredSignalWithMutexBase<
FEXCore::ForkableSharedMutex,
&FEXCore::ForkableSharedMutex::lock_shared,
&FEXCore::ForkableSharedMutex::unlock_shared>;
using ScopedDeferredSignalWithForkableUniqueLock = ScopedDeferredSignalWithMutexBase<
FEXCore::ForkableSharedMutex,
&FEXCore::ForkableSharedMutex::lock,
&FEXCore::ForkableSharedMutex::unlock>;
class ScopedSignalMasker final {
public:
ScopedSignalMasker() = default;
void Mask(uint64_t Mask) {
#ifndef _WIN32
// Mask all signals, storing the original incoming mask
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
#endif
}
// Move-only type
ScopedSignalMasker(const ScopedSignalMasker&) = delete;
ScopedSignalMasker& operator=(ScopedSignalMasker&) = delete;
ScopedSignalMasker(ScopedSignalMasker &&rhs) = default;
ScopedSignalMasker& operator=(ScopedSignalMasker &&) = default;
void Unmask() {
#ifndef _WIN32
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
#endif
}
private:
#ifndef _WIN32
uint64_t OriginalMask{};
#endif
};
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
class ScopedPotentialDeferredSignalWithMutexBase final {
public:
ScopedPotentialDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL)
: Mutex {&_Mutex}
, Thread {Thread} {
if (Thread) {
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
}
else {
Masker.Mask(Mask);
}
// Lock the mutex
(Mutex->*lock_fn)();
}
// No copy or assignment possible
ScopedPotentialDeferredSignalWithMutexBase(const ScopedPotentialDeferredSignalWithMutexBase&) = delete;
ScopedPotentialDeferredSignalWithMutexBase& operator=(ScopedPotentialDeferredSignalWithMutexBase&) = delete;
// Only move
ScopedPotentialDeferredSignalWithMutexBase(ScopedPotentialDeferredSignalWithMutexBase &&rhs)
: Mutex {rhs.Mutex}
, Thread {rhs.Thread} {
rhs.Mutex = nullptr;
}
~ScopedPotentialDeferredSignalWithMutexBase() {
if (Mutex != nullptr) {
// Unlock the mutex
(Mutex->*unlock_fn)();
if (Thread) {
#ifdef _M_X86_64
// Needs to be atomic so that operations can't end up getting reordered around this.
// Without this, the refcount and the signal access could get reordered.
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
// X86-64 must do an additional check around the store.
if ((Result - 1) == 0) {
// Must happen after the refcount store
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
}
#else
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
#endif
}
else {
// Unmask back to the original signal mask
Masker.Unmask();
}
}
}
private:
MutexType *Mutex;
ScopedSignalMasker Masker;
FEXCore::Core::InternalThreadState *Thread;
};
using ScopedPotentialDeferredSignalWithMutex = ScopedPotentialDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
using ScopedPotentialDeferredSignalWithSharedLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
using ScopedPotentialDeferredSignalWithUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
// Forkable variant
using ScopedPotentialDeferredSignalWithForkableMutex = ScopedPotentialDeferredSignalWithMutexBase<
FEXCore::ForkableUniqueMutex,
&FEXCore::ForkableUniqueMutex::lock,
&FEXCore::ForkableUniqueMutex::unlock>;
using ScopedPotentialDeferredSignalWithForkableSharedLock = ScopedPotentialDeferredSignalWithMutexBase<
FEXCore::ForkableSharedMutex,
&FEXCore::ForkableSharedMutex::lock_shared,
&FEXCore::ForkableSharedMutex::unlock_shared>;
using ScopedPotentialDeferredSignalWithForkableUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<
FEXCore::ForkableSharedMutex,
&FEXCore::ForkableSharedMutex::lock,
&FEXCore::ForkableSharedMutex::unlock>;
}
@@ -1,243 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Debug/InternalThreadState.h>
#include <atomic>
#include <cstdint>
#include <mutex>
#include <optional>
#include <signal.h>
#ifndef _WIN32
#include <sys/syscall.h>
#endif
#include <unistd.h>
#include <variant>
namespace FEXCore {
#ifndef _WIN32
// Replacement for std::mutexes to deal with unlocking issues in the face of Linux fork() semantics.
//
// A fork() only clones the parent's calling thread. Other threads are silently dropped, which permanently leaves any mutexes owned by them locked.
// To address this issue, ForkableUniqueMutex and ForkableSharedMutex provide a way to forcefully remove any dangling locks and reset the mutexes to their default state.
class ForkableUniqueMutex final {
public:
ForkableUniqueMutex()
: Mutex (PTHREAD_MUTEX_INITIALIZER) {
}
// Move-only type
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = default;
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = default;
void lock() {
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
}
void unlock() {
[[maybe_unused]] const auto Result = pthread_mutex_unlock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
}
// Initialize the internal pthread object to its default initializer state.
// Should only ever be used in the child process when a Linux fork() has occured.
void StealAndDropActiveLocks() {
Mutex = PTHREAD_MUTEX_INITIALIZER;
}
private:
pthread_mutex_t Mutex;
};
class ForkableSharedMutex final {
public:
ForkableSharedMutex()
: Mutex (PTHREAD_RWLOCK_INITIALIZER) {
}
// Move-only type
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
ForkableSharedMutex(ForkableSharedMutex &&rhs) = default;
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = default;
void lock() {
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
}
void unlock() {
[[maybe_unused]] const auto Result = pthread_rwlock_unlock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
}
void lock_shared() {
[[maybe_unused]] const auto Result = pthread_rwlock_rdlock(&Mutex);
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
}
void unlock_shared() {
unlock();
}
bool try_lock() {
const auto Result = pthread_rwlock_trywrlock(&Mutex);
return Result == 0;
}
bool try_lock_shared() {
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
return Result == 0;
}
// Initialize the internal pthread object to its default initializer state.
// Should only ever be used in the child process when a Linux fork() has occured.
void StealAndDropActiveLocks() {
Mutex = PTHREAD_RWLOCK_INITIALIZER;
}
private:
pthread_rwlock_t Mutex;
};
#else
// Windows doesn't support forking, so these can be standard mutexes.
class ForkableUniqueMutex final : public std::mutex {
public:
void StealAndDropActiveLocks() {
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
}
};
class ForkableSharedMutex final : public std::shared_mutex {
public:
void StealAndDropActiveLocks() {
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
}
};
#endif
// Helper class to manage deferred signal refcounting within a block scope
class DeferredSignalRefCountGuard final {
public:
explicit DeferredSignalRefCountGuard(FEXCore::Core::InternalThreadState *Thread) : Thread(Thread) {
// Needs to be atomic so that operations can't end up getting reordered around this.
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
}
// Move-only type
DeferredSignalRefCountGuard(const DeferredSignalRefCountGuard&) = delete;
DeferredSignalRefCountGuard& operator=(DeferredSignalRefCountGuard&) = delete;
DeferredSignalRefCountGuard(DeferredSignalRefCountGuard&& rhs) : Thread(rhs.Thread) {
rhs.Thread = nullptr;
}
~DeferredSignalRefCountGuard() {
if (Thread) {
#ifdef _M_X86_64
// Needs to be atomic so that operations can't end up getting reordered around this.
// Without this, the refcount and the signal access could get reordered.
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
// X86-64 must do an additional check around the store.
if ((Result - 1) == 0) {
// Must happen after the refcount store
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
}
#else
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
#endif
}
}
private:
FEXCore::Core::InternalThreadState *Thread;
};
#ifndef _WIN32
// Helper class to mask POSIX signals within a block scope
class ScopedSignalMasker final {
public:
explicit ScopedSignalMasker(uint64_t Mask) : OriginalMask(0) {
// Mask all signals, storing the original incoming mask
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &*OriginalMask, sizeof(*OriginalMask));
}
// Move-only type
ScopedSignalMasker(const ScopedSignalMasker&) = delete;
ScopedSignalMasker& operator=(ScopedSignalMasker&) = delete;
ScopedSignalMasker(ScopedSignalMasker&& rhs) : OriginalMask(rhs.OriginalMask) {
rhs.OriginalMask.reset();
}
~ScopedSignalMasker() {
if (OriginalMask) {
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(*OriginalMask));
}
}
private:
std::optional<uint64_t> OriginalMask{};
};
#endif
/**
* @brief Produces a wrapper object around a scoped lock of the given mutex
* while ensuring POSIX signals are masked while the mutex is locked
*
* Use this to prevent reentrancy issues of C++ mutexes with certain signal handlers.
* Common examples of such issues are:
* - C++ mutexes not unlocking due to a signal handler calling longjmp from within a scope owning the mutex
* - The signal handler itself using a mutex that would be re-locked if the handler gets invoked
* again before unlocking
*
* Ownership of the returned object may be moved, but it is NOT SAFE to move across threads.
*/
template<template<typename> class LockType = std::unique_lock, typename MutexType>
[[nodiscard]] static auto MaskSignalsAndLockMutex(MutexType& mutex, uint64_t Mask = ~0ULL) {
#ifndef _WIN32
// Signals are masked first, and then the lock is acquired
struct {
ScopedSignalMasker mask;
LockType<MutexType> lock;
} scope_guard { ScopedSignalMasker { Mask }, LockType<MutexType> { mutex } };
return scope_guard;
#else
// TODO: Doesn't block signals which may or may not cause issues.
return LockType<MutexType> { mutex };
#endif
}
/**
* @brief Produces a wrapper object around a scoped lock of the given mutex
* while bumping the Thread's deferred signal refcount while the mutex is
* locked.
*/
template<template<typename> class LockType = std::unique_lock, typename MutexType>
[[nodiscard]] static auto GuardSignalDeferringSection(MutexType& mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL) {
// Refcount is incremented first, and then the lock is acquired.
struct {
std::optional<DeferredSignalRefCountGuard> refcount;
LockType<MutexType> lock;
} scope_guard = { DeferredSignalRefCountGuard { Thread }, LockType<MutexType> { mutex } };
return scope_guard;
}
// Like GuardSignalDeferringSection but falls back to masking signals when Thread is nullptr
template<template<typename> class LockType = std::unique_lock, typename MutexType>
[[nodiscard]] static auto GuardSignalDeferringSectionWithFallback(MutexType& mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL) {
#ifndef _WIN32
using ExtraGuard = std::variant<ScopedSignalMasker, DeferredSignalRefCountGuard>;
#else
using ExtraGuard = std::variant<std::monostate, DeferredSignalRefCountGuard>;
#endif
struct {
ExtraGuard refcount_or_mask;
LockType<MutexType> lock;
} scope_guard {
Thread ? ExtraGuard { DeferredSignalRefCountGuard { Thread } }
#ifndef _WIN32
: ExtraGuard { ScopedSignalMasker { Mask } }
#else
: ExtraGuard { }
#endif
};
scope_guard.lock = LockType<MutexType> { mutex };
return scope_guard;
}
}
-6
View File
@@ -1712,12 +1712,6 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Evaluate into flags") {
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Carry flag invert") {
TEST_SINGLE(cfinv(), "cfinv");
}
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Arm to eXternal FLAG") {
TEST_SINGLE(axflag(), "axflag");
}
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: eXternal to Arm FLAG") {
TEST_SINGLE(xaflag(), "xaflag");
}
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Conditional compare - register") {
TEST_SINGLE(ccmn(Size::i32Bit, Reg::r29, Reg::r28, StatusFlags::None, Condition::CC_AL), "ccmn w29, w28, #nzcv, al");
TEST_SINGLE(ccmn(Size::i32Bit, Reg::r29, Reg::r28, StatusFlags::Flag_N, Condition::CC_AL), "ccmn w29, w28, #Nzcv, al");
@@ -0,0 +1,123 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/DeferredSignalMutex.h>
#include <atomic>
#include <cstdint>
#include <mutex>
#include <shared_mutex>
#ifndef _WIN32
#include <signal.h>
#include <sys/syscall.h>
#endif
#include <unistd.h>
namespace FHU {
/**
* @brief A drop-in replacement for std::lock_guard that masks POSIX signals while the mutex is locked
*
* Use this class to prevent reentrancy issues of C++ mutexes with certain signal handlers.
* Common examples of such issues are:
* - C++ mutexes not unlocking due to a signal handler longjmping out of a scope owning the mutex
* - The signal handler itself using a mutex that would be re-locked if the handler gets invoked
* again before unlocking
*
* Ownership of this object may be moved, but it is NOT SAFE to move across threads.
*
* Constructor order:
* 1) Mask signals
* 2) Lock Mutex
*
* Destructor Order:
* 1) Unlock Mutex
* 2) Unmask signals
*/
#ifndef _WIN32
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
class ScopedSignalMaskWithMutexBase final {
public:
ScopedSignalMaskWithMutexBase(MutexType &_Mutex, uint64_t Mask = ~0ULL)
: Mutex {&_Mutex} {
// Mask all signals, storing the original incoming mask
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
// Lock the mutex
(Mutex->*lock_fn)();
}
// No copy or assignment possible
ScopedSignalMaskWithMutexBase(const ScopedSignalMaskWithMutexBase&) = delete;
ScopedSignalMaskWithMutexBase& operator=(ScopedSignalMaskWithMutexBase&) = delete;
// Only move
ScopedSignalMaskWithMutexBase(ScopedSignalMaskWithMutexBase &&rhs)
: OriginalMask {rhs.OriginalMask}, Mutex {rhs.Mutex} {
rhs.Mutex = nullptr;
}
~ScopedSignalMaskWithMutexBase() {
if (Mutex != nullptr) {
// Unlock the mutex
(Mutex->*unlock_fn)();
// Unmask back to the original signal mask
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
}
}
private:
uint64_t OriginalMask{};
MutexType *Mutex;
};
#else
// TODO: Doesn't block signals which may or may not cause issues.
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
class ScopedSignalMaskWithMutexBase final {
public:
ScopedSignalMaskWithMutexBase(MutexType &_Mutex, [[maybe_unused]] uint64_t Mask = ~0ULL)
: Mutex {&_Mutex} {
// Lock the mutex
(Mutex->*lock_fn)();
}
// No copy or assignment possible
ScopedSignalMaskWithMutexBase(const ScopedSignalMaskWithMutexBase&) = delete;
ScopedSignalMaskWithMutexBase& operator=(ScopedSignalMaskWithMutexBase&) = delete;
// Only move
ScopedSignalMaskWithMutexBase(ScopedSignalMaskWithMutexBase &&rhs)
: Mutex {rhs.Mutex} {
rhs.Mutex = nullptr;
}
~ScopedSignalMaskWithMutexBase() {
if (Mutex != nullptr) {
// Unlock the mutex
(Mutex->*unlock_fn)();
}
}
private:
MutexType *Mutex;
};
#endif
using ScopedSignalMaskWithMutex = ScopedSignalMaskWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
using ScopedSignalMaskWithSharedLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
using ScopedSignalMaskWithUniqueLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
using ScopedSignalMaskWithForkableMutex = ScopedSignalMaskWithMutexBase<
FEXCore::ForkableUniqueMutex,
&FEXCore::ForkableUniqueMutex::lock,
&FEXCore::ForkableUniqueMutex::unlock>;
using ScopedSignalMaskWithForkableSharedLock = ScopedSignalMaskWithMutexBase<
FEXCore::ForkableSharedMutex,
&FEXCore::ForkableSharedMutex::lock_shared,
&FEXCore::ForkableSharedMutex::unlock_shared>;
using ScopedSignalMaskWithForkableUniqueLock = ScopedSignalMaskWithMutexBase<
FEXCore::ForkableSharedMutex,
&FEXCore::ForkableSharedMutex::lock,
&FEXCore::ForkableSharedMutex::unlock>;
}
+15 -4
View File
@@ -14,12 +14,14 @@ logger.setLevel(logging.ERROR)
@dataclass
class TestData:
name: str
optimal: int
expectedinstructioncount: int
code: bytes
instructions: list
def __init__(self, Name, ExpectedInstructionCount, Code, Instructions):
def __init__(self, Name, Optimal, ExpectedInstructionCount, Code, Instructions):
self.name = Name
self.expectedinstructioncount = ExpectedInstructionCount
self.optimal = Optimal
self.code = Code
self.instructions = Instructions
@@ -27,6 +29,10 @@ class TestData:
def Name(self):
return self.name
@property
def Optimal(self):
return self.optimal
@property
def ExpectedInstructionCount(self):
return self.expectedinstructioncount
@@ -52,7 +58,6 @@ class HostFeatures(Flag) :
FEATURE_RPRES = (1 << 7)
FEATURE_FLAGM = (1 << 8)
FEATURE_FLAGM2 = (1 << 9)
FEATURE_CRYPTO = (1 << 10)
HostFeaturesLookup = {
"SVE128" : HostFeatures.FEATURE_SVE128,
@@ -65,7 +70,6 @@ HostFeaturesLookup = {
"RPRES" : HostFeatures.FEATURE_RPRES,
"FLAGM" : HostFeatures.FEATURE_FLAGM,
"FLAGM2" : HostFeatures.FEATURE_FLAGM2,
"CRYPTO" : HostFeatures.FEATURE_CRYPTO,
}
def GetHostFeatures(data):
@@ -108,10 +112,15 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
for key, items in json_data["Instructions"].items():
ExpectedInstructionCount = 0
Optimal = 0
Instructions = []
if ("ExpectedInstructionCount" in items):
ExpectedInstructionCount = int(items["ExpectedInstructionCount"])
if ("Optimal" in items):
if items["Optimal"].upper() == "YES":
Optimal = 1
if ("Skip" in items):
if items["Skip"].upper() == "YES":
continue
@@ -154,7 +163,7 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
with open(tmp_asm_out, "rb") as tmp_asm_out_file:
binary_hex = tmp_asm_out_file.read()
TestDataMap[TestName] = TestData(key, ExpectedInstructionCount, binary_hex, Instructions)
TestDataMap[TestName] = TestData(key, Optimal, ExpectedInstructionCount, binary_hex, Instructions)
os.remove(tmp_asm)
os.remove(tmp_asm_out)
@@ -172,6 +181,7 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
# };
# struct TestInfo {
# char InstName[128];
# uint64_t Optimal;
# int64_t ExpectedInstructionCount;
# uint64_t CodeSize;
# uint64_t x86InstCount;
@@ -198,6 +208,7 @@ def parse_json_data(json_filepath, json_filename, json_data, output_binary_path)
# Add each test
for key, item in TestDataMap.items():
MemData += struct.pack('128s', item.Name.encode("ascii"))
MemData += struct.pack('Q', item.Optimal)
MemData += struct.pack('q', item.ExpectedInstructionCount)
MemData += struct.pack('Q', len(item.Code))
MemData += struct.pack('Q', len(item.Instructions))
+11 -12
View File
@@ -104,17 +104,6 @@ namespace FEXServerClient {
return GetServerLockFolder() + "RootFS.lock";
}
fextl::string GetTempFolder() {
auto XDGRuntimeEnv = getenv("XDG_RUNTIME_DIR");
if (XDGRuntimeEnv) {
// If the XDG runtime directory works then use that.
return XDGRuntimeEnv;
}
// Fallback to `/tmp/` if XDG_RUNTIME_DIR doesn't exist.
// Might not be ideal but we don't have much of a choice.
return fextl::string{std::filesystem::temp_directory_path().string()};
}
fextl::string GetServerMountFolder() {
// We need a FEXServer mount directory that has some tricky requirements.
// - We don't want to use `/tmp/` if possible.
@@ -130,7 +119,17 @@ namespace FEXServerClient {
// - If this path doesn't exist then fallback to `/tmp/` as a last resort.
// - pressure-vessel explicitly creates an internal XDG_RUNTIME_DIR inside its chroot.
// - This is okay since pressure-vessel rbinds the FEX rootfs from the host to `/run/pressure-vessel/interpreter-root`.
auto Folder = GetTempFolder();
fextl::string Folder{};
auto XDGRuntimeEnv = getenv("XDG_RUNTIME_DIR");
if (XDGRuntimeEnv) {
// If the XDG runtime directory works then use that.
Folder = XDGRuntimeEnv;
}
else {
// Fallback to `/tmp/` if XDG_RUNTIME_DIR doesn't exist.
// Might not be ideal but we don't have much of a choice.
Folder = std::filesystem::temp_directory_path().string();
}
if (FEXCore::Config::FindContainer() == "pressure-vessel") {
// In pressure-vessel the mount point changes location.
-1
View File
@@ -50,7 +50,6 @@ namespace FEXServerClient {
fextl::string GetServerLockFolder();
fextl::string GetServerLockFile();
fextl::string GetServerRootFSLockFile();
fextl::string GetTempFolder();
fextl::string GetServerMountFolder();
fextl::string GetServerSocketName();
int GetServerFD();
+6 -9
View File
@@ -227,6 +227,7 @@ void AssertHandler(char const *Message) {
struct TestInfo {
char TestInst[128];
uint64_t Optimal;
int64_t ExpectedInstructionCount;
uint64_t CodeSize;
uint64_t x86InstCount;
@@ -275,8 +276,8 @@ static bool TestInstructions(FEXCore::Context::Context *CTX, FEXCore::Core::Inte
LogMan::Msg::IFmt("Testing instruction '{}': {} host instructions", CurrentTest->TestInst, INSTStats->first.HostCodeInstructions);
// Show the code if the count of instructions changed to something we didn't expect.
bool ShouldShowCode =
// Show the code if we know the implementation isn't optimal or if the count of instructions changed to something we didn't expect.
bool ShouldShowCode = CurrentTest->Optimal == 0 ||
INSTStats->first.HostCodeInstructions != CurrentTest->ExpectedInstructionCount;
if (ShouldShowCode) {
@@ -466,7 +467,6 @@ int main(int argc, char **argv, char **const envp) {
FEATURE_RPRES = (1U << 7),
FEATURE_FLAGM = (1U << 8),
FEATURE_FLAGM2 = (1U << 9),
FEATURE_CRYPTO = (1U << 10),
};
uint64_t SVEWidth = 0;
@@ -503,12 +503,11 @@ int main(int argc, char **argv, char **const envp) {
if (TestHeaderData->EnabledHostFeatures & FEATURE_FLAGM2) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEFLAGM2);
}
if (TestHeaderData->EnabledHostFeatures & FEATURE_CRYPTO) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLECRYPTO);
}
// Always enable ARMv8.1 LSE atomics.
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEATOMICS);
// Always enable crypto extensions.
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLECRYPTO);
if (TestHeaderData->DisabledHostFeatures & FEATURE_SVE128) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLESVE);
@@ -540,9 +539,6 @@ int main(int argc, char **argv, char **const envp) {
if (TestHeaderData->DisabledHostFeatures & FEATURE_FLAGM2) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEFLAGM2);
}
if (TestHeaderData->DisabledHostFeatures & FEATURE_CRYPTO) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLECRYPTO);
}
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_HOSTFEATURES, fextl::fmt::format("{}", HostFeatureControl));
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_FORCESVEWIDTH, fextl::fmt::format("{}", SVEWidth));
@@ -553,6 +549,7 @@ int main(int argc, char **argv, char **const envp) {
// Create FEXCore context.
auto CTX = FEXCore::Context::Context::CreateNewContext();
CTX->InitializeContext();
auto SignalDelegation = FEX::DummyHandlers::CreateSignalDelegator();
auto SyscallHandler = FEX::DummyHandlers::CreateSyscallHandler();
+12 -2
View File
@@ -34,11 +34,21 @@ class DummySignalDelegator final : public FEXCore::SignalDelegator, public FEXCo
}
protected:
// Called from the thunk handler to handle the signal
void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *Info, void *UContext) override {}
void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) override;
void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) override;
private:
FEXCore::Core::InternalThreadState *GetTLSThread();
FEXCore::Core::InternalThreadState *GetTLSThread() override;
/**
* @brief Registers a signal handler for the host to handle a signal
*
* It's a process level signal handler so one must be careful
*/
void FrontendRegisterHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override {}
void FrontendRegisterFrontendHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override {}
};
fextl::unique_ptr<FEXCore::HLE::SyscallHandler> CreateSyscallHandler();
+2 -1
View File
@@ -106,7 +106,8 @@ void AOTGenSection(FEXCore::Context::Context *CTX, ELFCodeLoader::LoadedSection
setpriority(PRIO_PROCESS, FHU::Syscalls::gettid(), 19);
// Setup thread - Each compilation thread uses its own backing FEX thread
auto Thread = CTX->CreateThread(0, 0);
FEXCore::Core::CPUState state;
auto Thread = CTX->CreateThread(&state, FHU::Syscalls::gettid());
fextl::set<uint64_t> ExternalBranchesLocal;
CTX->ConfigureAOTGen(Thread, &ExternalBranchesLocal, SectionMaxAddress);
@@ -143,10 +143,6 @@ static inline uint64_t GetArmReg(void* ucontext, uint32_t id) {
return GetMContext(ucontext)->regs[id];
}
static inline uint64_t GetArmPState(void* ucontext) {
return GetMContext(ucontext)->pstate;
}
static inline uint64_t *GetArmGPRs(void* ucontext) {
return reinterpret_cast<uint64_t*>(GetMContext(ucontext)->regs);
}
@@ -317,10 +313,6 @@ static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
ERROR_AND_DIE_FMT("Not implemented for x86 host");
}
static inline uint64_t GetArmPState(void* ucontext) {
ERROR_AND_DIE_FMT("Not implemented for x86 host");
}
static inline uint64_t *GetArmGPRs(void* ucontext) {
ERROR_AND_DIE_FMT("Not implemented for x86 host");
}
+3 -11
View File
@@ -12,7 +12,6 @@ $end_info$
#include "Common/Config.h"
#include "ELFCodeLoader.h"
#include "VDSO_Emulation.h"
#include "LinuxSyscalls/GdbServer.h"
#include "LinuxSyscalls/LinuxAllocator.h"
#include "LinuxSyscalls/Syscalls.h"
#include "LinuxSyscalls/Utils/Threads.h"
@@ -444,6 +443,7 @@ int main(int argc, char **argv, char **const envp) {
FEXCore::Context::InitializeStaticTables(Loader.Is64BitMode() ? FEXCore::Context::MODE_64BIT : FEXCore::Context::MODE_32BIT);
auto CTX = FEXCore::Context::Context::CreateNewContext();
CTX->InitializeContext();
// Setup TSO hardware emulation immediately after initializing the context.
FEX::TSO::SetupTSOEmulation(CTX.get());
@@ -474,14 +474,7 @@ int main(int argc, char **argv, char **const envp) {
CTX->SetSignalDelegator(SignalDelegation.get());
CTX->SetSyscallHandler(SyscallHandler.get());
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
fextl::unique_ptr<FEX::GdbServer> DebugServer;
if (GdbServer) {
DebugServer = fextl::make_unique<FEX::GdbServer>(CTX.get(), SignalDelegation.get(), SyscallHandler.get());
}
auto ParentThread = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
// Pass in our VDSO thunks
CTX->AppendThunkDefinitions(FEX::VDSO::GetVDSOThunkDefinitions());
@@ -557,9 +550,8 @@ int main(int argc, char **argv, char **const envp) {
}
}
auto ProgramStatus = ParentThread->StatusCode;
auto ProgramStatus = CTX->GetProgramStatus();
DebugServer.reset();
SyscallHandler.reset();
SignalDelegation.reset();
+6 -2
View File
@@ -160,6 +160,7 @@ int main(int argc, char **argv, char **const envp)
FEXCore::Context::InitializeStaticTables();
auto CTX = FEXCore::Context::Context::CreateNewContext();
CTX->InitializeContext();
auto SignalDelegation = FEX::HLE::CreateSignalDelegator(CTX.get(), {});
@@ -178,7 +179,7 @@ int main(int argc, char **argv, char **const envp)
if (Loader.LoadIR(CTX.get()))
{
auto ParentThread = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
auto ShutdownReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
@@ -210,7 +211,10 @@ int main(int argc, char **argv, char **const envp)
LogMan::Msg::DFmt("Reason we left VM: {}", FEXCore::ToUnderlying(ShutdownReason));
// Just re-use compare state. It also checks against the expected values in config.
const bool Passed = Loader.CompareStates(&ParentThread->CurrentFrame->State, SupportsAVX);
FEXCore::Core::CPUState State;
CTX->GetCPUState(&State);
const bool Passed = Loader.CompareStates(&State, SupportsAVX);
LogMan::Msg::IFmt("Passed? {}\n", Passed ? "Yes" : "No");
@@ -1,7 +1,6 @@
add_compile_options(-fno-operator-names)
set (SRCS
GdbServer.cpp
EmulatedFiles/EmulatedFiles.cpp
FileManagement.cpp
LinuxAllocator.cpp
@@ -14,7 +14,6 @@ $end_info$
#include <FEXCore/Core/CodeLoader.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Utils/CPUInfo.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/string.h>
@@ -47,22 +46,9 @@ namespace FEX::EmulatedFile {
*
* @return A temporary file that we can use
*/
static int GenTmpFD(const char *pathname, int flags) {
uint32_t memfd_flags {MFD_ALLOW_SEALING};
if (flags & O_CLOEXEC) memfd_flags |= MFD_CLOEXEC;
return memfd_create(pathname, memfd_flags);
}
// Seal the tmpfd features by sealing them all.
// Makes the tmpfd read-only.
static void SealTmpFD(int fd) {
fcntl(fd, F_ADD_SEALS,
F_SEAL_SEAL |
F_SEAL_SHRINK |
F_SEAL_GROW |
F_SEAL_WRITE |
F_SEAL_FUTURE_WRITE);
static int GenTmpFD() {
int fd = open("/tmp", O_RDWR | O_TMPFILE | O_EXCL, S_IRUSR | S_IWUSR);
return fd;
}
fextl::string GenerateCPUInfo(FEXCore::Context::Context *ctx, uint32_t CPUCores) {
@@ -635,23 +621,21 @@ namespace FEX::EmulatedFile {
}
EmulatedFDManager::EmulatedFDManager(FEXCore::Context::Context *ctx)
: CTX {ctx}
, ThreadsConfig { FEXCore::CPUInfo::CalculateNumberOfCPUs() } {
: CTX {ctx} {
FDReadCreators["/proc/cpuinfo"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
// Only allow a single thread to initialize the cpu_info.
// Jit in-case multiple threads try to initialize at once.
// Check if deferred cpuinfo initialization has occured.
std::call_once(cpu_info_initialized, [&]() { cpu_info = GenerateCPUInfo(ctx, ThreadsConfig); });
std::call_once(cpu_info_initialized, [&]() { cpu_info = GenerateCPUInfo(ctx, ThreadsConfig()); });
int FD = GenTmpFD(pathname, flags);
int FD = GenTmpFD();
write(FD, (void*)&cpu_info.at(0), cpu_info.size());
lseek(FD, 0, SEEK_SET);
SealTmpFD(FD);
return FD;
};
FDReadCreators["/proc/sys/kernel/osrelease"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
int FD = GenTmpFD(pathname, flags);
int FD = GenTmpFD();
uint32_t GuestVersion = FEX::HLE::_SyscallHandler->GetGuestKernelVersion();
char Tmp[64]{};
snprintf(Tmp, sizeof(Tmp), "%d.%d.%d\n",
@@ -661,12 +645,11 @@ namespace FEX::EmulatedFile {
// + 1 to ensure null at the end
write(FD, Tmp, strlen(Tmp) + 1);
lseek(FD, 0, SEEK_SET);
SealTmpFD(FD);
return FD;
};
FDReadCreators["/proc/version"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
int FD = GenTmpFD(pathname, flags);
int FD = GenTmpFD();
// UTS version NEEDS to be in a format that can pass to `date -d`
// Format of this is Linux version <Release> (<Compile By>@<Compile Host>) (<Linux Compiler>) #<version> {SMP, PREEMPT, PREEMPT_RT} <UTS version>\n"
const char kernel_version[] = "Linux version %d.%d.%d (FEX@FEX) (clang) #" GIT_DESCRIBE_STRING " SMP " __DATE__ " " __TIME__ "\n";
@@ -679,15 +662,13 @@ namespace FEX::EmulatedFile {
// + 1 to ensure null at the end
write(FD, Tmp, strlen(Tmp) + 1);
lseek(FD, 0, SEEK_SET);
SealTmpFD(FD);
return FD;
};
auto NumCPUCores = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
int FD = GenTmpFD(pathname, flags);
int FD = GenTmpFD();
write(FD, (void*)&cpus_online.at(0), cpus_online.size());
lseek(FD, 0, SEEK_SET);
SealTmpFD(FD);
return FD;
};
@@ -700,7 +681,7 @@ namespace FEX::EmulatedFile {
FDReadCreators["/proc/self/auxv"] = &EmulatedFDManager::ProcAuxv;
auto cmdline_handler = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
int FD = GenTmpFD(pathname, flags);
int FD = GenTmpFD();
auto CodeLoader = FEX::HLE::_SyscallHandler->GetCodeLoader();
auto Args = CodeLoader->GetApplicationArguments();
char NullChar{};
@@ -714,7 +695,6 @@ namespace FEX::EmulatedFile {
// One additional null terminator to finish the list
lseek(FD, 0, SEEK_SET);
SealTmpFD(FD);
return FD;
};
@@ -722,8 +702,9 @@ namespace FEX::EmulatedFile {
fextl::string procCmdLine = fextl::fmt::format("/proc/{}/cmdline", getpid());
FDReadCreators[procCmdLine] = cmdline_handler;
if (ThreadsConfig > 1) {
cpus_online = fextl::fmt::format("0-{}", ThreadsConfig - 1);
uint64_t CPUCores = ThreadsConfig();
if (CPUCores > 1) {
cpus_online = fextl::fmt::format("0-{}", CPUCores - 1);
}
else {
cpus_online = "0";
@@ -740,7 +721,6 @@ namespace FEX::EmulatedFile {
auto Creator = FDReadCreators.end();
if (pathname) {
Creator = FDReadCreators.find(pathname);
Path = pathname;
}
if (Creator == FDReadCreators.end()) {
@@ -808,10 +788,9 @@ namespace FEX::EmulatedFile {
return -1;
}
int FD = GenTmpFD(pathname, flags);
int FD = GenTmpFD();
write(FD, (void*)auxvBase, auxvSize);
lseek(FD, 0, SEEK_SET);
SealTmpFD(FD);
return FD;
}
}
@@ -34,6 +34,6 @@ namespace FEX::EmulatedFile {
fextl::unordered_map<fextl::string, FDReadStringFunc> FDReadCreators;
static int32_t ProcAuxv(FEXCore::Context::Context* ctx, int32_t fd, const char* pathname, int32_t flags, mode_t mode);
const uint32_t ThreadsConfig;
FEX_CONFIG_OPT(ThreadsConfig, THREADS);
};
}
@@ -150,38 +150,6 @@ namespace FEX::HLE {
return SigInfoLayout::LAYOUT_KILL;
}
void SignalDelegator::HandleSignal(int Signal, void *Info, void *UContext) {
// Let the host take first stab at handling the signal
auto Thread = GetTLSThread();
if (!Thread) {
LogMan::Msg::AFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
}
else {
SignalHandler &Handler = HostHandlers[Signal];
for (auto &HandlerFunc : Handler.Handlers) {
if (HandlerFunc(Thread, Signal, Info, UContext)) {
// If the host handler handled the fault then we can continue now
return;
}
}
if (Handler.FrontendHandler &&
Handler.FrontendHandler(Thread, Signal, Info, UContext)) {
return;
}
// Now let the frontend handle the signal
// It's clearly a guest signal and this ends up being an OS specific issue
HandleGuestSignal(Thread, Signal, Info, UContext);
}
}
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
SetHostSignalHandler(Signal, Func, Required);
FrontendRegisterHostSignalHandler(Signal, Func, Required);
}
void SignalDelegator::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
#ifdef _M_ARM_64
for (size_t i = 0; i < Config.SRAGPRCount; i++) {
@@ -1169,13 +1137,12 @@ namespace FEX::HLE {
++Thread->CurrentFrame->SignalHandlerRefCounter;
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
const bool WasInJIT = Thread->CPUBackend->IsAddressInCodeBuffer(OldPC);
// Spill the SRA regardless of signal handler type
// We are going to be returning to the top of the dispatcher which will fill again
// Otherwise we might load garbage
if (Config.StaticRegisterAllocation) {
if (WasInJIT) {
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
uint32_t IgnoreMask{};
#ifdef _M_ARM_64
if (Frame->InSyscallInfo != 0) {
@@ -1240,7 +1207,7 @@ namespace FEX::HLE {
// Backup where we think the RIP currently is
ContextBackup->OriginalRIP = CTX->RestoreRIPFromHostPC(Thread, ArchHelpers::Context::GetPc(ucontext));
// Calculate eflags upfront.
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(Thread, WasInJIT, ArchHelpers::Context::GetArmGPRs(ucontext), ArchHelpers::Context::GetArmPState(ucontext));
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(Thread);
if (Is64BitMode) {
NewGuestSP = SetupFrame_x64(Thread, ContextBackup, Frame, Signal, HostSigInfo, ucontext, GuestAction, GuestStack, NewGuestSP, eflags);
@@ -1843,7 +1810,7 @@ namespace FEX::HLE {
ThreadData.Thread = nullptr;
}
void SignalDelegator::FrontendRegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
void SignalDelegator::FrontendRegisterHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) {
// Linux signal handlers are per-process rather than per thread
// Multiple threads could be calling in to this
std::lock_guard lk(HostDelegatorMutex);
@@ -1851,7 +1818,7 @@ namespace FEX::HLE {
InstallHostThunk(Signal);
}
void SignalDelegator::FrontendRegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
void SignalDelegator::FrontendRegisterFrontendHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) {
// Linux signal handlers are per-process rather than per thread
// Multiple threads could be calling in to this
std::lock_guard lk(HostDelegatorMutex);
@@ -40,36 +40,20 @@ namespace FEX::HLE {
class SignalDelegator final : public FEXCore::SignalDelegator, public FEXCore::Allocator::FEXAllocOperators {
public:
constexpr static size_t MAX_SIGNALS {64};
// Use the last signal just so we are less likely to ever conflict with something that the guest application is using
// 64 is used internally by Valgrind
constexpr static size_t SIGNAL_FOR_PAUSE {63};
// Returns true if the host handled the signal
// Arguments are the same as sigaction handler
SignalDelegator(FEXCore::Context::Context *_CTX, const std::string_view ApplicationName);
~SignalDelegator() override;
// Called from the signal trampoline function.
void HandleSignal(int Signal, void *Info, void *UContext);
void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) override;
void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Registers a signal handler for the host to handle a signal
*
* It's a process level signal handler so one must be careful
*/
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
/**
* @brief Registers a signal handler for the host to handle a signal specifically for guest handling
*
* It's a process level signal handler so one must be careful
*/
void RegisterHostSignalHandlerForGuest(int Signal, HostSignalDelegatorFunctionForGuest Func);
void RegisterHostSignalHandlerForGuest(int Signal, FEX::HLE::HostSignalDelegatorFunctionForGuest Func);
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
/**
@@ -123,27 +107,21 @@ namespace FEX::HLE {
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
void SaveTelemetry();
private:
FEXCore::Core::InternalThreadState *GetTLSThread();
protected:
// Called from the thunk handler to handle the signal
void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *Info, void *UContext);
void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *Info, void *UContext) override;
FEXCore::Core::InternalThreadState *GetTLSThread() override;
/**
* @brief Registers a signal handler for the host to handle a signal
*
* It's a process level signal handler so one must be careful
*/
void FrontendRegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
void FrontendRegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
void SetHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
HostHandlers[Signal].Handlers.push_back(std::move(Func));
}
void SetFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
HostHandlers[Signal].FrontendHandler = std::move(Func);
}
void FrontendRegisterHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override;
void FrontendRegisterFrontendHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override;
private:
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(Core, CORE);
fextl::string const ApplicationName;
@@ -177,10 +155,6 @@ namespace FEX::HLE {
FEX::HLE::HostSignalDelegatorFunctionForGuest GuestHandler{};
GuestSigAction GuestAction{};
DefaultBehaviour DefaultBehaviour {DEFAULT_TERM};
// Callbacks
fextl::vector<HostSignalDelegatorFunction> Handlers{};
HostSignalDelegatorFunction FrontendHandler{};
};
std::array<SignalHandler, MAX_SIGNALS + 1> HostHandlers{};
@@ -525,14 +525,6 @@ static uint64_t Clone3Handler(FEXCore::Core::CpuStateFrame *Frame, FEX::HLE::clo
uint64_t CloneHandler(FEXCore::Core::CpuStateFrame *Frame, FEX::HLE::clone3_args *args) {
uint64_t flags = args->args.flags;
if (flags & CLONE_CLEAR_SIGHAND) {
// CLONE_CLEAR_SIGHAND was added in kernel 5.5. FEX doesn't properly support this.
// glibc started using this flag in 2.38 as an optimization for posix_spawn.
// If clone returns EINVAL or ENOSYS then it will fallback to the non-optimized path.
LogMan::Msg::IFmt("CLONE_CLEAR_SIGHAND passed to clone3. Returning EINVAL.");
return -EINVAL;
}
auto HasUnhandledFlags = [](FEX::HLE::clone3_args *args) -> bool {
constexpr uint64_t UNHANDLED_FLAGS =
CLONE_NEWNS |
@@ -16,7 +16,7 @@ $end_info$
#include <FEXCore/HLE/SourcecodeResolver.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/Utils/DeferredSignalMutex.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/memory.h>
@@ -175,6 +175,7 @@ public:
FEX_CONFIG_OPT(IsInterpreterInstalled, INTERPRETER_INSTALLED);
FEX_CONFIG_OPT(Filename, APP_FILENAME);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
FEX_CONFIG_OPT(ThreadsConfig, THREADS);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
@@ -89,8 +89,21 @@ namespace FEX::HLE {
REGISTER_SYSCALL_IMPL_FLAGS(getcpu, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
[](FEXCore::Core::CpuStateFrame *Frame, unsigned *cpu, unsigned *node, struct getcpu_cache *tcache) -> uint64_t {
uint32_t LocalCPU{};
uint32_t LocalNode{};
// tcache is ignored
uint64_t Result = ::syscall(SYSCALL_DEF(getcpu), cpu, node, nullptr);
uint64_t Result = ::syscall(SYSCALL_DEF(getcpu), cpu ? &LocalCPU : nullptr, node ? &LocalNode : nullptr, nullptr);
if (Result == 0) {
if (cpu) {
// Ensure we don't return a number over our number of emulated cores
*cpu = LocalCPU % FEX::HLE::_SyscallHandler->ThreadsConfig();
}
if (node) {
// Just claim we are part of node zero
*node = 0;
}
}
SYSCALL_ERRNO();
});
@@ -80,14 +80,35 @@ namespace FEX::HLE {
REGISTER_SYSCALL_IMPL_FLAGS(sched_setaffinity, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY | SyscallFlags::NOSIDEEFFECTS,
[](FEXCore::Core::CpuStateFrame *Frame, pid_t pid, size_t cpusetsize, const unsigned long *mask) -> uint64_t {
uint64_t Result = ::syscall(SYSCALL_DEF(sched_setaffinity), pid, cpusetsize, mask);
SYSCALL_ERRNO();
return 0;
});
REGISTER_SYSCALL_IMPL_FLAGS(sched_getaffinity, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
[](FEXCore::Core::CpuStateFrame *Frame, pid_t pid, size_t cpusetsize, unsigned char *mask) -> uint64_t {
uint64_t Result = ::syscall(SYSCALL_DEF(sched_getaffinity), pid, cpusetsize, mask);
SYSCALL_ERRNO();
uint64_t Cores = FEX::HLE::_SyscallHandler->ThreadsConfig();
// Bytes need to round up to size of uint64_t
uint64_t Bytes = FEXCore::AlignUp(Cores, sizeof(uint64_t));
// cpusetsize needs to be 8byte aligned
if (cpusetsize & (sizeof(uint64_t) - 1)) {
return -EINVAL;
}
// If we don't have enough bytes to store the resulting structure
// then we need to return -EINVAL
if (cpusetsize < Bytes) {
return -EINVAL;
}
memset(mask, 0, Bytes);
for (uint64_t i = 0; i < Cores; ++i) {
mask[i / 8] |= (1 << (i % 8));
}
// Returns the number of bytes written in to mask
return Bytes;
});
REGISTER_SYSCALL_IMPL_PASS_FLAGS(sched_setattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
@@ -58,7 +58,7 @@ namespace FEX::HLE {
NewThreadState.gregs[FEXCore::X86State::REG_RSP] = args->args.stack;
}
auto NewThread = CTX->CreateThread(0, 0, &NewThreadState, args->args.parent_tid);
auto NewThread = CTX->CreateThread(&NewThreadState, args->args.parent_tid);
CTX->InitializeThread(NewThread);
if (FEX::HLE::_SyscallHandler->Is64BitMode()) {
@@ -131,7 +131,7 @@ namespace FEX::HLE {
}
// Overwrite thread
NewThread = CTX->CreateThread(0, 0, &NewThreadState, GuestArgs->parent_tid);
NewThread = CTX->CreateThread(&NewThreadState, GuestArgs->parent_tid);
// CLONE_PARENT_SETTID, CLONE_CHILD_SETTID, CLONE_CHILD_CLEARTID, CLONE_PIDFD will be handled by kernel
// Call execution thread directly since we already are on the new thread
@@ -16,10 +16,11 @@ $end_info$
#include "LinuxSyscalls/Syscalls.h"
#include <FEXHeaderUtils/TypeDefines.h>
#include <FEXHeaderUtils/ScopedSignalMask.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/Utils/DeferredSignalMutex.h>
namespace FEX::HLE {
@@ -54,7 +55,7 @@ bool SyscallHandler::HandleSegfault(FEXCore::Core::InternalThreadState *Thread,
{
// Can't use the deferred signal lock in the SIGSEGV handler.
auto lk = FEXCore::MaskSignalsAndLockMutex<std::shared_lock>(_SyscallHandler->VMATracking.Mutex);
FHU::ScopedSignalMaskWithForkableSharedLock lk(_SyscallHandler->VMATracking.Mutex);
auto VMATracking = &_SyscallHandler->VMATracking;
@@ -111,7 +112,7 @@ void SyscallHandler::MarkGuestExecutableRange(FEXCore::Core::InternalThreadState
return;
}
auto lk = FEXCore::GuardSignalDeferringSection<std::shared_lock>(VMATracking.Mutex, Thread);
FEXCore::ScopedDeferredSignalWithForkableSharedLock lk(VMATracking.Mutex, Thread);
// Find the first mapping at or after the range ends, or ::end().
// Top points to the address after the end of the range
@@ -166,7 +167,7 @@ void SyscallHandler::MarkGuestExecutableRange(FEXCore::Core::InternalThreadState
// Used for AOT
FEXCore::HLE::AOTIRCacheEntryLookupResult SyscallHandler::LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestAddr) {
auto lk = FEXCore::GuardSignalDeferringSection<std::shared_lock>(VMATracking.Mutex, Thread);
FEXCore::ScopedDeferredSignalWithForkableSharedLock lk(VMATracking.Mutex, Thread);
// Get the first mapping after GuestAddr, or end
// GuestAddr is inclusive
@@ -193,8 +194,8 @@ void SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState *Thread, uintp
{
// NOTE: Frontend calls this with a nullptr Thread during initialization, but
// providing this code with a valid Thread object earlier would allow
// us to be more optimal by using GuardSignalDeferringSection instead
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(VMATracking.Mutex, Thread);
// us to be more optimal by using ScopedDeferredSignalWithUniqueLock instead
FEXCore::ScopedPotentialDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
static uint64_t AnonSharedId = 1;
@@ -243,9 +244,9 @@ void SyscallHandler::TrackMunmap(FEXCore::Core::InternalThreadState *Thread, uin
{
// Frontend calls this with nullptr Thread during initialization.
// This is why `GuardSignalDeferringSectionWithFallback` is used here.
// This is why `ScopedPotentialDeferredSignalWithUniqueLock` is used here.
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(VMATracking.Mutex, Thread);
FEXCore::ScopedPotentialDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
VMATracking.ClearUnsafe(CTX, Base, Size);
}
@@ -259,7 +260,7 @@ void SyscallHandler::TrackMprotect(FEXCore::Core::InternalThreadState *Thread, u
Size = FEXCore::AlignUp(Size, FHU::FEX_PAGE_SIZE);
{
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
VMATracking.ChangeUnsafe(Base, Size, VMAProt::fromProt(Prot));
}
@@ -274,7 +275,7 @@ void SyscallHandler::TrackMremap(FEXCore::Core::InternalThreadState *Thread, uin
NewSize = FEXCore::AlignUp(NewSize, FHU::FEX_PAGE_SIZE);
{
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
const auto OldVMA = VMATracking.LookupVMAUnsafe(OldAddress);
@@ -332,7 +333,7 @@ void SyscallHandler::TrackShmat(FEXCore::Core::InternalThreadState *Thread, int
uint64_t Length = stat.shm_segsz;
{
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
// TODO
MRID mrid{SpecialDev::SHM, static_cast<uint64_t>(shmid)};
@@ -354,7 +355,7 @@ void SyscallHandler::TrackShmat(FEXCore::Core::InternalThreadState *Thread, int
void SyscallHandler::TrackShmdt(FEXCore::Core::InternalThreadState *Thread, uintptr_t Base) {
uintptr_t Length = 0;
{
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
Length = VMATracking.ClearShmUnsafe(CTX, Base);
}
@@ -368,7 +369,7 @@ void SyscallHandler::TrackShmdt(FEXCore::Core::InternalThreadState *Thread, uint
void SyscallHandler::TrackMadvise(FEXCore::Core::InternalThreadState *Thread, uintptr_t Base, uintptr_t Size, int advice) {
Size = FEXCore::AlignUp(Size, FHU::FEX_PAGE_SIZE);
{
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
FEXCore::ScopedDeferredSignalWithForkableUniqueLock lk(VMATracking.Mutex, Thread);
// TODO
}
}
+5 -3
View File
@@ -245,6 +245,8 @@ int main(int argc, char **argv, char **const envp) {
auto CTX = FEXCore::Context::Context::CreateNewContext();
CTX->InitializeContext();
#ifndef _WIN32
auto SignalDelegation = FEX::HLE::CreateSignalDelegator(CTX.get(), {});
#else
@@ -301,9 +303,9 @@ int main(int argc, char **argv, char **const envp) {
CTX->SetSignalDelegator(SignalDelegation.get());
CTX->SetSyscallHandler(SyscallHandler.get());
auto ParentThread = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
bool Result1 = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
if (!ParentThread) {
if (!Result1) {
return 1;
}
@@ -313,7 +315,7 @@ int main(int argc, char **argv, char **const envp) {
}
// Just re-use compare state. It also checks against the expected values in config.
memcpy(&State, &ParentThread->CurrentFrame->State, sizeof(State));
CTX->GetCPUState(&State);
SyscallHandler.reset();
}
+52 -124
View File
@@ -212,7 +212,6 @@ namespace WorkingAppsTester {
// EroFS specific
static bool Has_EroFSFuse {false};
static bool Has_EroFSFsck {false};
void CheckCurl() {
// Check if curl exists on the host
@@ -296,23 +295,11 @@ namespace WorkingAppsTester {
Has_EroFSFuse = Result != -1;
}
void CheckEroFSFsck() {
std::vector<const char*> ExecveArgs = {
"fsck.erofs",
"-V",
nullptr,
};
int32_t Result = Exec::ExecAndWaitForResponseRedirect(ExecveArgs[0], const_cast<char* const*>(ExecveArgs.data()), -1, -1);
Has_EroFSFsck = Result != -1;
}
void Init() {
CheckCurl();
CheckSquashfuse();
CheckUnsquashfs();
CheckEroFSFuse();
CheckEroFSFsck();
}
}
@@ -489,9 +476,12 @@ namespace WebFileFetcher {
const static std::string DownloadURL = "https://rootfs.fex-emu.gg/RootFS_links.json";
std::string DownloadToString(const std::string &URL) {
std::string BigArgs =
fmt::format("curl {}", URL);
std::vector<const char*> ExecveArgs = {
"curl",
URL.c_str(),
"/bin/sh",
"-c",
BigArgs.c_str(),
nullptr,
};
@@ -502,11 +492,12 @@ namespace WebFileFetcher {
auto filename = URL.substr(URL.find_last_of('/') + 1);
auto PathName = Path + filename;
std::string BigArgs =
fmt::format("curl {} -o {}", URL, PathName);
std::vector<const char*> ExecveArgs = {
"curl",
URL.c_str(),
"-o",
PathName.c_str(),
"/bin/sh",
"-c",
BigArgs.c_str(),
nullptr,
};
@@ -1100,7 +1091,7 @@ namespace UnSquash {
bool Extract = true;
std::error_code ec;
if (std::filesystem::exists(TargetFolder, ec)) {
fextl::string Question = "Target folder \"" + FolderName + "\" already exists. Overwrite?";
fextl::string Question = FolderName + " Already exists. Overwrite?";
if (AskForConfirmation(Question)) {
if (std::filesystem::remove_all(TargetFolder, ec) != ~0ULL) {
Extract = true;
@@ -1123,40 +1114,6 @@ namespace UnSquash {
return false;
}
bool ExtractEroFS(const fextl::string &Path, const fextl::string &RootFS, const fextl::string &FolderName) {
auto TargetFolder = Path + FolderName;
bool Extract = true;
std::error_code ec;
if (std::filesystem::exists(TargetFolder, ec)) {
fextl::string Question = "Target folder \"" + FolderName + "\" already exists. Overwrite?";
if (AskForConfirmation(Question)) {
if (std::filesystem::remove_all(TargetFolder, ec) != ~0ULL) {
Extract = true;
}
if (ec) {
ExecWithInfo("Couldn't remove previous directory. Won't extract.");
}
}
}
if (Extract) {
ExecWithInfo("Extracting Erofs. This might take a few minutes.");
const auto ExtractOption = fmt::format("--extract={}", TargetFolder);
const std::vector<const char*> ExecveArgs = {
"fsck.erofs",
ExtractOption.c_str(),
RootFS.c_str(),
nullptr,
};
return Exec::ExecAndWaitForResponse(ExecveArgs[0], const_cast<char* const*>(ExecveArgs.data())) == 0;
}
return false;
}
}
int main(int argc, char **argv, char **const envp) {
@@ -1283,91 +1240,62 @@ int main(int argc, char **argv, char **const envp) {
}
}
struct ExtractStrings {
char const *ExtractOrAsIs;
char const *AsIsSinceMounterNonFunctional;
char const *AsIsSinceExtractorNonFunctional;
char const *AsIsSinceNothingWorks;
};
ArgOptions::CompressedImageOption UseImageAs {ArgOptions::CompressedUsageOption};
bool HasExtractor{};
bool HasMounter{};
std::function<bool (const fextl::string &Path, const fextl::string &RootFS, const fextl::string &FolderName)> ExtractHelper;
ExtractStrings ExtractingStrings;
if (Target.Type == WebFileFetcher::FileTargets::FileType::TYPE_SQUASHFS) {
HasExtractor = WorkingAppsTester::Has_Unsquashfs;
HasMounter = WorkingAppsTester::Has_Squashfuse;
ExtractHelper = UnSquash::UnsquashRootFS;
ExtractingStrings =
{
"Do you wish to extract the squashfs file or use it as-is?",
"Squashfuse doesn't work. Do you wish to extract the squashfs file?",
"Unsquashfs doesn't work. Do you want to use the squashfs file as-is?",
"Unsquashfs and squashfuse isn't working. Leaving rootfs as-is",
};
}
else if (Target.Type == WebFileFetcher::FileTargets::FileType::TYPE_EROFS) {
HasExtractor = WorkingAppsTester::Has_EroFSFsck;
HasMounter = WorkingAppsTester::Has_EroFSFuse;
ExtractHelper = UnSquash::ExtractEroFS;
ExtractingStrings =
{
"Do you wish to extract the erofs file or use it as-is?",
"erofsfuse doesn't work. Do you wish to extract the erofs file?",
"Extracting erofs doesn't work. Do you want to use the erofs file as-is?",
"Extracting erofs and erofsfuse isn't working. Leaving rootfs as-is",
};
}
int32_t Result{};
std::vector<fextl::string> Args = {
"Extract",
"As-Is",
};
if (UseImageAs == ArgOptions::CompressedImageOption::OPTION_ASK) {
if (HasExtractor) {
if (HasMounter) {
Result = AskForConfirmationList(ExtractingStrings.ExtractOrAsIs, Args);
if (Result == 0) {
UseImageAs = ArgOptions::CompressedImageOption::OPTION_EXTRACT;
ArgOptions::CompressedImageOption UseImageAs {ArgOptions::CompressedUsageOption};
if (Target.Type == WebFileFetcher::FileTargets::FileType::TYPE_SQUASHFS) {
int32_t Result{};
if (UseImageAs == ArgOptions::CompressedImageOption::OPTION_ASK) {
if (WorkingAppsTester::Has_Unsquashfs) {
if (WorkingAppsTester::Has_Squashfuse) {
Result = AskForConfirmationList("Do you wish to extract the squashfs file or use it as-is?", Args);
if (Result == 0) {
UseImageAs = ArgOptions::CompressedImageOption::OPTION_EXTRACT;
}
else if (Result == 1) {
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
}
}
else if (Result == 1) {
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
else {
Args.pop_back();
Result = AskForConfirmationList("Squashfuse doesn't work. Do you wish to extract the squashfs file?", Args);
if (Result == 0) {
UseImageAs = ArgOptions::CompressedImageOption::OPTION_EXTRACT;
}
}
}
else {
Args.pop_back();
Result = AskForConfirmationList(ExtractingStrings.AsIsSinceMounterNonFunctional, Args);
if (Result == 0) {
UseImageAs = ArgOptions::CompressedImageOption::OPTION_EXTRACT;
if (WorkingAppsTester::Has_Squashfuse) {
Args.erase(Args.begin());
Result = AskForConfirmationList("Unsquashfs doesn't work. Do you want to use the squashfs file as-is?", Args);
if (Result == 0) {
// We removed an argument, Just change "As-Is" from 0 to 1 for later logic to work
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
}
}
else {
Args.erase(Args.begin());
ExecWithInfo("Unsquashfs and squashfuse isn't working. Leaving rootfs as-is");
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
}
}
}
else {
if (HasMounter) {
Args.erase(Args.begin());
Result = AskForConfirmationList(ExtractingStrings.AsIsSinceExtractorNonFunctional, Args);
if (Result == 0) {
// We removed an argument, Just change "As-Is" from 0 to 1 for later logic to work
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
}
}
else {
Args.erase(Args.begin());
ExecWithInfo(ExtractingStrings.AsIsSinceNothingWorks);
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
if (UseImageAs == ArgOptions::CompressedImageOption::OPTION_EXTRACT) {
auto FolderName = filename.substr(0, filename.find_last_of('.'));
if (UnSquash::UnsquashRootFS(RootFS, PathName, FolderName)) {
// Remove the .sqsh suffix since we extracted to that
filename = FolderName;
}
}
}
if (UseImageAs == ArgOptions::CompressedImageOption::OPTION_EXTRACT) {
auto FolderName = filename.substr(0, filename.find_last_of('.'));
if (ExtractHelper(RootFS, PathName, FolderName)) {
// Remove the image file suffix since we extracted to that.
filename = FolderName;
}
else if (Target.Type == WebFileFetcher::FileTargets::FileType::TYPE_EROFS) {
// Once erofs tooling is available for easy extraction, offer the same settings to extract as squashfs.
// Currently this is unavailable, would need a mount + copy + unmount dance.
UseImageAs = ArgOptions::CompressedImageOption::OPTION_ASIS;
}
if (AskForConfirmation("Do you wish to set this RootFS as default?")) {
+9 -10
View File
@@ -22,13 +22,12 @@ EXPORTS
Wow64PassExceptionToGuest @16
Wow64PrepareForDebuggerAttach @17 PRIVATE
Wow64PrepareForException @18
Wow64ProcessPendingCrossProcessItems @19
Wow64RaiseException @20
Wow64ShallowThunkAllocObjectAttributes32TO64_FNC @21 PRIVATE
Wow64ShallowThunkAllocSecurityQualityOfService32TO64_FNC @22 PRIVATE
Wow64ShallowThunkSIZE_T32TO64 @23 PRIVATE
Wow64ShallowThunkSIZE_T64TO32 @24 PRIVATE
Wow64SuspendLocalThread @25 PRIVATE
Wow64SystemServiceEx @26
Wow64ValidateUserCallTarget @27 PRIVATE
Wow64ValidateUserCallTargetFilter @28 PRIVATE
Wow64RaiseException @19
Wow64ShallowThunkAllocObjectAttributes32TO64_FNC @20 PRIVATE
Wow64ShallowThunkAllocSecurityQualityOfService32TO64_FNC @21 PRIVATE
Wow64ShallowThunkSIZE_T32TO64 @22 PRIVATE
Wow64ShallowThunkSIZE_T64TO32 @23 PRIVATE
Wow64SuspendLocalThread @24
Wow64SystemServiceEx @25
Wow64ValidateUserCallTarget @26 PRIVATE
Wow64ValidateUserCallTargetFilter @27 PRIVATE
+3 -23
View File
@@ -35,7 +35,6 @@ $end_info$
#include <atomic>
#include <mutex>
#include <utility>
#include <unordered_set>
#include <ntstatus.h>
#include <windef.h>
#include <winternl.h>
@@ -95,7 +94,6 @@ namespace {
SYSTEM_CPU_INFORMATION CpuInfo{};
std::mutex ThreadSuspendLock;
std::unordered_set<DWORD> InitializedWOWThreads; // Set of TIDs, `ThreadSuspendLock` must be locked when accessing
std::pair<NTSTATUS, TLS> GetThreadTLS(HANDLE Thread) {
THREAD_BASIC_INFORMATION Info;
@@ -181,7 +179,7 @@ namespace Context {
Context->Esp = State.gregs[FEXCore::X86State::REG_RSP];
Context->Eip = State.rip;
Context->EFlags = CTX->ReconstructCompactedEFLAGS(Thread, false, nullptr, 0);
Context->EFlags = CTX->ReconstructCompactedEFLAGS(Thread);
Context->SegEs = State.es_idx;
Context->SegCs = State.cs_idx;
@@ -462,7 +460,6 @@ public:
const uint64_t EntryRAX = Frame->State.gregs[FEXCore::X86State::REG_RAX];
Context::UnlockJITContext();
Wow64ProcessPendingCrossProcessItems();
ReturnRAX = static_cast<uint64_t>(Wow64SystemServiceEx(static_cast<UINT>(EntryRAX),
reinterpret_cast<UINT *>(ReturnRSP + 4)));
Context::LockJITContext();
@@ -515,6 +512,7 @@ void BTCpuProcessInit() {
SyscallHandler = fextl::make_unique<WowSyscallHandler>();
CTX = FEXCore::Context::Context::CreateNewContext();
CTX->InitializeContext();
CTX->SetSignalDelegator(SignalDelegator.get());
CTX->SetSyscallHandler(SyscallHandler.get());
CTX->InitCore(0, 0);
@@ -554,10 +552,8 @@ void BTCpuProcessInit() {
}
NTSTATUS BTCpuThreadInit() {
GetTLS().ThreadState() = CTX->CreateThread(0, 0);
GetTLS().ThreadState() = CTX->CreateThread(nullptr, 0);
std::scoped_lock Lock(ThreadSuspendLock);
InitializedWOWThreads.emplace(GetCurrentThreadId());
return STATUS_SUCCESS;
}
@@ -567,17 +563,6 @@ NTSTATUS BTCpuThreadTerm(HANDLE Thread) {
return Err;
}
{
THREAD_BASIC_INFORMATION Info;
if (NTSTATUS Err = NtQueryInformationThread(Thread, ThreadBasicInformation, &Info, sizeof(Info), nullptr); Err) {
return Err;
}
const auto ThreadTID = reinterpret_cast<uint64_t>(Info.ClientId.UniqueThread);
std::scoped_lock Lock(ThreadSuspendLock);
InitializedWOWThreads.erase(ThreadTID);
}
CTX->DestroyThread(TLS.ThreadState());
return STATUS_SUCCESS;
}
@@ -680,11 +665,6 @@ NTSTATUS BTCpuSuspendLocalThread(HANDLE Thread, ULONG *Count) {
}
std::scoped_lock Lock(ThreadSuspendLock);
// If the thread hasn't yet been initialized, suspend it without special handling as it wont yet have entered the JIT
if (!InitializedWOWThreads.contains(ThreadTID))
return NtSuspendThread(Thread, Count);
// If CONTROL_IN_JIT is unset at this point, then it can never be set (and thus the JIT cannot be reentered) as
// CONTROL_PAUSED has been set, as such, while this may redundantly request interrupts in rare cases it will never
// miss them
-1
View File
@@ -90,7 +90,6 @@ typedef enum _MEMORY_INFORMATION_CLASS {
} MEMORY_INFORMATION_CLASS;
NTSTATUS WINAPI Wow64SystemServiceEx(UINT,UINT*);
void WINAPI Wow64ProcessPendingCrossProcessItems(void);
NTSTATUS WINAPI RtlWow64SetThreadContext(HANDLE,const WOW64_CONTEXT*);
NTSTATUS WINAPI RtlWow64GetThreadContext(HANDLE,WOW64_CONTEXT*);
+10 -2
View File
@@ -232,6 +232,14 @@ extern "C" {
return rv;
}
static void LockMutexFunction(LockInfoPtr) {
fprintf(stderr, "libX11: LockMutex\n");
}
static void UnlockMutexFunction(LockInfoPtr) {
fprintf(stderr, "libX11: LockMutex\n");
}
int XFree(void* ptr) {
// This function must be able to handle both guest heap pointers *and* host heap pointers,
// so it only forwards to the native host library for the latter.
@@ -360,8 +368,8 @@ extern "C" {
return fexfn_pack_XUnregisterIMInstantiateCallback(dpy, rdb, res_name, res_class, AllocateHostTrampolineForGuestFunction(callback), client_data);
}
void (*_XLockMutex_fn)(LockInfoPtr) = nullptr;
void (*_XUnlockMutex_fn)(LockInfoPtr) = nullptr;
void (*_XLockMutex_fn)(LockInfoPtr) = LockMutexFunction;
void (*_XUnlockMutex_fn)(LockInfoPtr) = UnlockMutexFunction;
LockInfoPtr _Xglobal_lock = (LockInfoPtr)0x4142434445464748ULL;
}
+5 -9
View File
@@ -1,4 +1,4 @@
# FEX-2312.1
# FEX-2311.1
## FEXCore
See [FEXCore/Readme.md](../FEXCore/Readme.md) for more details
@@ -64,10 +64,6 @@ Metadata that drives the frontend x86/64 decoding
- [X87.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp): Handles x86/64 x87 to IR
- [X87F64.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp): Handles x86/64 x87 to IR
- [OpcodeDispatcher.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
- [OpcodeDispatcher_BACKUP_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_BACKUP_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
- [OpcodeDispatcher_BASE_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_BASE_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
- [OpcodeDispatcher_LOCAL_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_LOCAL_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
- [OpcodeDispatcher_REMOTE_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_REMOTE_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
@@ -81,6 +77,10 @@ Logic that binds various parts together
Emulation mainloop related glue logic
- [Core.cpp](../FEXCore/Source/Interface/Core/Core.cpp): Glues Frontend, OpDispatcher and IR Opts & Compilation, LookupCache, Dispatcher and provides the Execution loop entrypoint
#### gdbserver
- [GdbServer.cpp](../FEXCore/Source/Interface/Core/GdbServer.cpp): Provides a gdb interface to the guest state
- [GdbServer.h](../FEXCore/Source/Interface/Core/GdbServer.h)
#### log-manager
- [LogManager.cpp](../FEXCore/Source/Utils/LogManager.cpp)
@@ -143,10 +143,6 @@ Text -> IR
- [X87.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp): Handles x86/64 x87 to IR
- [X87F64.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp): Handles x86/64 x87 to IR
- [OpcodeDispatcher.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
- [OpcodeDispatcher_BACKUP_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_BACKUP_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
- [OpcodeDispatcher_BASE_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_BASE_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
- [OpcodeDispatcher_LOCAL_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_LOCAL_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
- [OpcodeDispatcher_REMOTE_124790.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher_REMOTE_124790.cpp): Handles x86/64 ops to IR, no-pf opt, local-flags opt
## ThunkLibs
See [ThunkLibs/README.md](../ThunkLibs/README.md) for more details
@@ -1,51 +0,0 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0x000000007dbf2800",
"RDX": "0x0000000000000000",
"RBX": "0x000000000000004f",
"RCX": "0x000000000000004f",
"RBP": "0x0000000000009e4f",
"RSI": "0x0000000000009e4f",
"RSP": "0x000000000000004f"
},
"Mode": "32BIT"
}
%endif
; FEX had a bug where smaller than 64-bit imul could leave garbage data in the upper 32-bits of the 32-bit result.
; This would cause subsequent instructions after the imul to receive garbage bits.
; In particular this would feed in to address calculation in DXVK with "Dungeon Defenders" doing address calculation.
; The address calculation did something similar to:
; xor edx, edx
; mov eax, 0x7dbf2800
; imul ebx, ebx, 0xaaaaaaab
; div ebx
; Divide expected 0x4f but received 0xffffffb1'0000'004f
; Dividend
xor edx, edx
mov eax, 0x7dbf2800
; Multiply starting value
mov ebx, 0xED
jmp .test
.test:
; imul 1-src
mov edi, 0xaaaaaaab
imul di, bx
mov esp, 0xaaaaaaab
imul esp, ebx
; imul 2-src 8-bit check
imul bp, bx, 0xab
imul esi, ebx, 0xab
; imul 2-src 16-bit check
imul cx, bx, 0xaaab
imul ebx, ebx, 0xaaaaaaab
hlt
-45
View File
@@ -1,45 +0,0 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0x4",
"RBX": "0xFFFFFFFFFFFFFFF4",
"RCX": "0x0",
"RDX": "0x1337"
}
}
%endif
; FEX had a bug where bzhi would fail to update SF. Test that bzhi correctly
; sets ZF/SF correctly based on the result.
mov rcx, 4
mov rbx, -12
; Result is 0x4
bzhi rax, rbx, rcx
mov rdx, 0xdead1
jz .fail
mov rdx, 0xdead2
js .fail
; Result is -12
mov rcx, 64
bzhi rdx, rbx, rcx
mov rdx, 0xdead3
jz .fail
mov rdx, 0xdead4
jns .fail
; Result is 0x00
mov rdx, 0
bzhi rcx, rbx, rdx
mov rdx, 0xdead5
jnz .fail
mov rdx, 0xdead6
js .fail
mov rdx, 0x1337
hlt
.fail:
hlt
@@ -1,15 +0,0 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0x500000020"
}
}
%endif
; FEX had a bug in its `TestNZ` opcode where it would try to load a constant in to the tst instruction
; If the constant didn't fit in a logical encoding it would generate invalid instructions and also crash.
; This snippet of code was found in libGLX.so.0.0.0 when trying to load steamwebhelper.
mov eax, 0x28000001
shl rax, 0x5
hlt
@@ -1,49 +0,0 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0",
"XMM0": ["0", "0"]
}
}
%endif
; FEX-Emu has a bug around NZCV flags getting spilled and filled.
; The bug comes down to NZCV actually being 32-bit but our IR incorrectly assumed that all flags were 8-bit.
; Once a spill situation happened, it would only store and reload the lower 8-bits of the NZCV flag which wasn't correct.
; This caused this code to infinite loop and read past memory and crash.
; Code found from Ender Lilies in their `sha1_block_data_order` function which is significantly longer than this snippit.
lea rsi, [rel .data_vecs]
mov rax, 1
; Break visibility
jmp loop_top
loop_top:
; Decrement counter.
dec rax
; Load rsi + 0x40 in to rbx
lea rbx, [rsi+0x40]
; Move rbx in to rsi, incrementing the pointer by 64-bytes if rax isn't zero.
cmovne rsi, rbx
; Do a sha1rnds4, which uses enough temporaries to spill NZCV which picks up a crash.
sha1rnds4 xmm0, xmm0, 0x0
; This memory access will crash once we loop too many times.
movdqu xmm0, [rsi]
; Jump back to the top
jne loop_top
hlt
.data_vecs:
dq 0, 0, 0, 0
dq 0, 0, 0, 0
dq 0, 0, 0, 0
dq 0, 0, 0, 0
dq 0, 0, 0, 0
dq 0, 0, 0, 0
@@ -12,6 +12,7 @@
"Instructions": {
"roundss xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"Nearest rounding",
"0x66 0x0f 0x3a 0x0a"
@@ -22,6 +23,7 @@
},
"roundss xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"-inf rounding",
"0x66 0x0f 0x3a 0x0a"
@@ -32,6 +34,7 @@
},
"roundss xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"+inf rounding",
"0x66 0x0f 0x3a 0x0a"
@@ -42,6 +45,7 @@
},
"roundss xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"truncate rounding",
"0x66 0x0f 0x3a 0x0a"
@@ -52,6 +56,7 @@
},
"roundss xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"host rounding mode rounding",
"0x66 0x0f 0x3a 0x0a"
@@ -62,6 +67,7 @@
},
"roundsd xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"Nearest rounding",
"0x66 0x0f 0x3a 0x0b"
@@ -72,6 +78,7 @@
},
"roundsd xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"-inf rounding",
"0x66 0x0f 0x3a 0x0b"
@@ -82,6 +89,7 @@
},
"roundsd xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"+inf rounding",
"0x66 0x0f 0x3a 0x0b"
@@ -92,6 +100,7 @@
},
"roundsd xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"truncate rounding",
"0x66 0x0f 0x3a 0x0b"
@@ -102,6 +111,7 @@
},
"roundsd xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"host rounding mode rounding",
"0x66 0x0f 0x3a 0x0b"
@@ -11,6 +11,7 @@
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0x0f 0x2a"
],
@@ -21,11 +22,12 @@
},
"cvtpi2ps xmm0, mm0": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #768]",
"ldr d2, [x28, #752]",
"scvtf v16.2s, v2.2s"
]
}
@@ -11,6 +11,7 @@
"Instructions": {
"cvtsi2ss xmm0, eax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
@@ -20,6 +21,7 @@
},
"cvtsi2ss xmm0, dword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
@@ -30,6 +32,7 @@
},
"cvtsi2ss xmm0, qword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
@@ -40,6 +43,7 @@
},
"sqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x51",
"ExpectedArm64ASM": [
"fsqrt s16, s17"
@@ -47,6 +51,7 @@
},
"rsqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"0xf3 0x0f 0x52"
@@ -59,6 +64,7 @@
},
"rcpss xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"0xf3 0x0f 0x53"
@@ -70,6 +76,7 @@
},
"addss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x58"
],
@@ -79,6 +86,7 @@
},
"mulss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x59"
],
@@ -88,6 +96,7 @@
},
"cvtss2sd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"fcvt d16, s17"
@@ -95,6 +104,7 @@
},
"cvtss2sd xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"ldr d2, [x4]",
@@ -103,6 +113,7 @@
},
"subss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5c"
],
@@ -112,6 +123,7 @@
},
"minss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5d"
],
@@ -121,6 +133,7 @@
},
"divss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5e"
],
@@ -130,6 +143,7 @@
},
"maxss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5f"
],
@@ -139,6 +153,7 @@
},
"cmpss xmm0, xmm1, 0": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -148,6 +163,7 @@
},
"cmpss xmm0, xmm1, 1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -157,6 +173,7 @@
},
"cmpss xmm0, xmm1, 2": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -166,6 +183,7 @@
},
"cmpss xmm0, xmm1, 3": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -180,6 +198,7 @@
},
"cmpss xmm0, xmm1, 4": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -192,6 +211,7 @@
},
"cmpss xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -203,6 +223,7 @@
},
"cmpss xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -214,6 +235,7 @@
},
"cmpss xmm0, xmm1, 7": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -11,6 +11,7 @@
"Instructions": {
"cvtsi2sd xmm0, eax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
@@ -20,6 +21,7 @@
},
"cvtsi2sd xmm0, dword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
@@ -30,6 +32,7 @@
},
"cvtsi2sd xmm0, rax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
@@ -39,6 +42,7 @@
},
"cvtsi2sd xmm0, qword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
@@ -49,6 +53,7 @@
},
"sqrtsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x51"
],
@@ -58,6 +63,7 @@
},
"addsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x58"
],
@@ -67,6 +73,7 @@
},
"mulsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x59"
],
@@ -76,6 +83,7 @@
},
"cvtsd2ss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5a"
],
@@ -85,6 +93,7 @@
},
"cvtsd2ss xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5a"
],
@@ -95,6 +104,7 @@
},
"subsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5c"
],
@@ -104,6 +114,7 @@
},
"minsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5d"
],
@@ -113,6 +124,7 @@
},
"divsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5e"
],
@@ -122,6 +134,7 @@
},
"maxsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5f"
],
@@ -131,6 +144,7 @@
},
"cmpsd xmm0, xmm1, 0": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -140,6 +154,7 @@
},
"cmpsd xmm0, xmm1, 1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -149,6 +164,7 @@
},
"cmpsd xmm0, xmm1, 2": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -158,6 +174,7 @@
},
"cmpsd xmm0, xmm1, 3": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -172,6 +189,7 @@
},
"cmpsd xmm0, xmm1, 4": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -184,6 +202,7 @@
},
"cmpsd xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -195,6 +214,7 @@
},
"cmpsd xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -206,6 +226,7 @@
},
"cmpsd xmm0, xmm1, 7": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -12,6 +12,7 @@
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0x0f 0x2a"
],
@@ -22,11 +23,12 @@
},
"cvtpi2ps xmm0, mm0": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #768]",
"ldr d2, [x28, #752]",
"scvtf v16.2s, v2.2s"
]
}
@@ -12,6 +12,7 @@
"Instructions": {
"cvtsi2ss xmm0, eax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
@@ -21,6 +22,7 @@
},
"cvtsi2ss xmm0, dword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
@@ -31,6 +33,7 @@
},
"cvtsi2ss xmm0, qword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
@@ -41,6 +44,7 @@
},
"sqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x51",
"ExpectedArm64ASM": [
"fsqrt s16, s17"
@@ -48,6 +52,7 @@
},
"rsqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"0xf3 0x0f 0x52"
@@ -60,6 +65,7 @@
},
"rcpss xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"0xf3 0x0f 0x53"
@@ -71,6 +77,7 @@
},
"addss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x58"
],
@@ -80,6 +87,7 @@
},
"mulss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x59"
],
@@ -89,6 +97,7 @@
},
"cvtss2sd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"fcvt d16, s17"
@@ -96,6 +105,7 @@
},
"cvtss2sd xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"ldr d2, [x4]",
@@ -104,6 +114,7 @@
},
"subss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5c"
],
@@ -113,6 +124,7 @@
},
"minss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5d"
],
@@ -122,6 +134,7 @@
},
"divss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5e"
],
@@ -131,6 +144,7 @@
},
"maxss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5f"
],
@@ -140,6 +154,7 @@
},
"cmpss xmm0, xmm1, 0": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -149,6 +164,7 @@
},
"cmpss xmm0, xmm1, 1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -158,6 +174,7 @@
},
"cmpss xmm0, xmm1, 2": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -167,6 +184,7 @@
},
"cmpss xmm0, xmm1, 3": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -180,6 +198,7 @@
},
"cmpss xmm0, xmm1, 4": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -191,6 +210,7 @@
},
"cmpss xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -202,6 +222,7 @@
},
"cmpss xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -213,6 +234,7 @@
},
"cmpss xmm0, xmm1, 7": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
@@ -12,6 +12,7 @@
"Instructions": {
"cvtsi2sd xmm0, eax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
@@ -21,6 +22,7 @@
},
"cvtsi2sd xmm0, dword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
@@ -31,6 +33,7 @@
},
"cvtsi2sd xmm0, rax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
@@ -40,6 +43,7 @@
},
"cvtsi2sd xmm0, qword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
@@ -50,6 +54,7 @@
},
"sqrtsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x51"
],
@@ -59,6 +64,7 @@
},
"addsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x58"
],
@@ -68,6 +74,7 @@
},
"mulsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x59"
],
@@ -77,6 +84,7 @@
},
"cvtsd2ss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5a"
],
@@ -86,6 +94,7 @@
},
"cvtsd2ss xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5a"
],
@@ -96,6 +105,7 @@
},
"subsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5c"
],
@@ -105,6 +115,7 @@
},
"minsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5d"
],
@@ -114,6 +125,7 @@
},
"divsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5e"
],
@@ -123,6 +135,7 @@
},
"maxsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5f"
],
@@ -132,6 +145,7 @@
},
"cmpsd xmm0, xmm1, 0": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -141,6 +155,7 @@
},
"cmpsd xmm0, xmm1, 1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -150,6 +165,7 @@
},
"cmpsd xmm0, xmm1, 2": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -159,6 +175,7 @@
},
"cmpsd xmm0, xmm1, 3": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -172,6 +189,7 @@
},
"cmpsd xmm0, xmm1, 4": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -183,6 +201,7 @@
},
"cmpsd xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -194,6 +213,7 @@
},
"cmpsd xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -205,6 +225,7 @@
},
"cmpsd xmm0, xmm1, 7": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
@@ -11,6 +11,7 @@
"Instructions": {
"vsqrtss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x51 128-bit"
],
@@ -21,6 +22,7 @@
},
"vsqrtsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x51 128-bit"
],
@@ -31,6 +33,7 @@
},
"vrsqrtss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"Map 1 0b10 0x52 128-bit"
@@ -44,6 +47,7 @@
},
"vrcpss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"Map 1 0b10 0x53 128-bit"
@@ -56,6 +60,7 @@
},
"vcmpss xmm0, xmm1, xmm2, 0x00": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
@@ -66,6 +71,7 @@
},
"vcmpss xmm0, xmm1, xmm2, 0x01": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
@@ -76,6 +82,7 @@
},
"vcmpss xmm0, xmm1, xmm2, 0x02": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
@@ -86,6 +93,7 @@
},
"vcmpss xmm0, xmm1, xmm2, 0x03": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
@@ -100,6 +108,7 @@
},
"vcmpss xmm0, xmm1, xmm2, 0x04": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
@@ -112,6 +121,7 @@
},
"vcmpss xmm0, xmm1, xmm2, 0x05": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
@@ -124,6 +134,7 @@
},
"vcmpss xmm0, xmm1, xmm2, 0x06": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
@@ -136,6 +147,7 @@
},
"vcmpss xmm0, xmm1, xmm2, 0x07": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
@@ -149,6 +161,7 @@
},
"vcmpsd xmm0, xmm1, xmm2, 0x00": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
@@ -159,6 +172,7 @@
},
"vcmpsd xmm0, xmm1, xmm2, 0x01": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
@@ -169,6 +183,7 @@
},
"vcmpsd xmm0, xmm1, xmm2, 0x02": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
@@ -179,6 +194,7 @@
},
"vcmpsd xmm0, xmm1, xmm2, 0x03": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
@@ -193,6 +209,7 @@
},
"vcmpsd xmm0, xmm1, xmm2, 0x04": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
@@ -205,6 +222,7 @@
},
"vcmpsd xmm0, xmm1, xmm2, 0x05": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
@@ -217,6 +235,7 @@
},
"vcmpsd xmm0, xmm1, xmm2, 0x06": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
@@ -229,6 +248,7 @@
},
"vcmpsd xmm0, xmm1, xmm2, 0x07": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
@@ -242,6 +262,7 @@
},
"vcvtsi2ss xmm0, xmm1, eax": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x2A 128-bit"
],
@@ -252,6 +273,7 @@
},
"vcvtsi2ss xmm0, xmm1, rax": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x2A 128-bit"
],
@@ -262,6 +284,7 @@
},
"vcvtsi2sd xmm0, xmm1, eax": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x2A 128-bit"
],
@@ -272,6 +295,7 @@
},
"vcvtsi2sd xmm0, xmm1, rax": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x2A 128-bit"
],
@@ -282,6 +306,7 @@
},
"vmulss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x59 128-bit"
],
@@ -292,6 +317,7 @@
},
"vmulsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x59 128-bit"
],
@@ -302,6 +328,7 @@
},
"vcvtss2sd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5a 128-bit"
],
@@ -312,6 +339,7 @@
},
"vcvtsd2ss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5a 128-bit"
],
@@ -322,6 +350,7 @@
},
"vsubss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5c 128-bit"
],
@@ -332,6 +361,7 @@
},
"vsubsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5c 128-bit"
],
@@ -342,6 +372,7 @@
},
"vminss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5d 128-bit"
],
@@ -352,6 +383,7 @@
},
"vminsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5d 128-bit"
],
@@ -362,6 +394,7 @@
},
"vdivss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5e 128-bit"
],
@@ -372,6 +405,7 @@
},
"vdivsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5e 128-bit"
],
@@ -382,6 +416,7 @@
},
"vmaxss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5f 128-bit"
],
@@ -392,6 +427,7 @@
},
"vmaxsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5f 128-bit"
],
@@ -402,6 +438,7 @@
},
"vminps xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"Map 1 0b00 0x5d 128-bit"
],
@@ -413,6 +450,7 @@
},
"vminps ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Comment": [
"Map 1 0b00 0x5d 256-bit"
],
@@ -426,6 +464,7 @@
},
"vminpd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"Map 1 0b01 0x5d 128-bit"
],
@@ -437,6 +476,7 @@
},
"vminpd ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Comment": [
"Map 1 0b01 0x5d 256-bit"
],
@@ -11,6 +11,7 @@
"Instructions": {
"vroundss xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"nearest rounding",
"Map 3 0b01 0x0a 128-bit"
@@ -22,6 +23,7 @@
},
"vroundss xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"-inf rounding",
"Map 3 0b01 0x0a 128-bit"
@@ -33,6 +35,7 @@
},
"vroundss xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"+inf rounding",
"Map 3 0b01 0x0a 128-bit"
@@ -44,6 +47,7 @@
},
"vroundss xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"truncate rounding",
"Map 3 0b01 0x0a 128-bit"
@@ -55,6 +59,7 @@
},
"vroundss xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"host mode rounding",
"Map 3 0b01 0x0a 128-bit"
@@ -66,6 +71,7 @@
},
"vroundsd xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"nearest rounding",
"Map 3 0b01 0x0b 128-bit"
@@ -77,6 +83,7 @@
},
"vroundsd xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"-inf rounding",
"Map 3 0b01 0x0b 128-bit"
@@ -88,6 +95,7 @@
},
"vroundsd xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"+inf rounding",
"Map 3 0b01 0x0b 128-bit"
@@ -99,6 +107,7 @@
},
"vroundsd xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"truncate rounding",
"Map 3 0b01 0x0b 128-bit"
@@ -110,6 +119,7 @@
},
"vroundsd xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"host mode rounding",
"Map 3 0b01 0x0b 128-bit"
File diff suppressed because it is too large. Load diff
@@ -1,138 +0,0 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"CRYPTO"
],
"DisabledHostFeatures": [
"SVE128",
"SVE256",
"AFP"
]
},
"Instructions": {
"sha1nexte xmm0, xmm1": {
"ExpectedInstructionCount": 6,
"Comment": [
"0x66 0x0f 0x38 0xc8"
],
"ExpectedArm64ASM": [
"dup v2.4s, v16.s[3]",
"unimplemented (Unimplemented)",
"dup v2.4s, v2.s[0]",
"add v2.4s, v17.4s, v2.4s",
"mov v16.16b, v17.16b",
"mov v16.s[3], v2.s[3]"
]
},
"sha256msg1 xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Comment": [
"0x66 0x0f 0x38 0xcc"
],
"ExpectedArm64ASM": [
"unimplemented (Unimplemented)"
]
},
"aesimc xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Comment": [
"0x66 0x0f 0x38 0xdb"
],
"ExpectedArm64ASM": [
"unimplemented (Unimplemented)"
]
},
"aesenc xmm0, xmm1": {
"ExpectedInstructionCount": 4,
"Comment": [
"0x66 0x0f 0x38 0xdc"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"unimplemented (Unimplemented)",
"unimplemented (Unimplemented)",
"eor v16.16b, v16.16b, v17.16b"
]
},
"aesenclast xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Comment": [
"0x66 0x0f 0x38 0xdd"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"unimplemented (Unimplemented)",
"eor v16.16b, v16.16b, v17.16b"
]
},
"aesdec xmm0, xmm1": {
"ExpectedInstructionCount": 4,
"Comment": [
"0x66 0x0f 0x38 0xde"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"unimplemented (Unimplemented)",
"unimplemented (Unimplemented)",
"eor v16.16b, v16.16b, v17.16b"
]
},
"aesdeclast xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Comment": [
"0x66 0x0f 0x38 0xdf"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"unimplemented (Unimplemented)",
"eor v16.16b, v16.16b, v17.16b"
]
},
"crc32 eax, bl": {
"ExpectedInstructionCount": 1,
"Comment": [
"0xf2 0x0f 0x38 0xf0"
],
"ExpectedArm64ASM": [
"crc32cb w4, w4, w7"
]
},
"crc32 eax, bx": {
"ExpectedInstructionCount": 1,
"Comment": [
"0xf2 0x0f 0x38 0xf1"
],
"ExpectedArm64ASM": [
"crc32ch w4, w4, w7"
]
},
"crc32 eax, ebx": {
"ExpectedInstructionCount": 1,
"Comment": [
"0xf2 0x0f 0x38 0xf1"
],
"ExpectedArm64ASM": [
"crc32cw w4, w4, w7"
]
},
"crc32 rax, bl": {
"ExpectedInstructionCount": 1,
"Comment": [
"0xf2 0x0f 0x38 0xf0"
],
"ExpectedArm64ASM": [
"crc32cb w4, w4, w7"
]
},
"crc32 rax, rbx": {
"ExpectedInstructionCount": 1,
"Comment": [
"0xf2 0x0f 0x38 0xf1"
],
"ExpectedArm64ASM": [
"crc32cx w4, w4, x7"
]
}
}
}
@@ -1,82 +0,0 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"CRYPTO"
],
"DisabledHostFeatures": [
"SVE128",
"SVE256",
"AFP"
]
},
"Instructions": {
"pclmulqdq xmm0, xmm1, 00000b": {
"ExpectedInstructionCount": 1,
"Comment": [
"0x66 0x0f 0x3a 0x44"
],
"ExpectedArm64ASM": [
"unallocated (Unallocated)"
]
},
"pclmulqdq xmm0, xmm1, 00001b": {
"ExpectedInstructionCount": 2,
"Comment": [
"0x66 0x0f 0x3a 0x44"
],
"ExpectedArm64ASM": [
"dup v0.2d, v16.d[1]",
"unallocated (Unallocated)"
]
},
"pclmulqdq xmm0, xmm1, 10000b": {
"ExpectedInstructionCount": 2,
"Comment": [
"0x66 0x0f 0x3a 0x44"
],
"ExpectedArm64ASM": [
"dup v0.2d, v17.d[1]",
"unallocated (Unallocated)"
]
},
"pclmulqdq xmm0, xmm1, 10001b": {
"ExpectedInstructionCount": 1,
"Comment": [
"0x66 0x0f 0x3a 0x44"
],
"ExpectedArm64ASM": [
"unallocated (Unallocated)"
]
},
"aeskeygenassist xmm0, xmm1, 0": {
"ExpectedInstructionCount": 5,
"Comment": [
"0x66 0x0f 0x3a 0xdf"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #2080]",
"movi v3.2d, #0x0",
"mov v16.16b, v17.16b",
"unimplemented (Unimplemented)",
"tbl v16.16b, {v16.16b}, v2.16b"
]
},
"aeskeygenassist xmm0, xmm1, 0xFF": {
"ExpectedInstructionCount": 8,
"Comment": [
"0x66 0x0f 0x3a 0xdf"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #2080]",
"movi v3.2d, #0x0",
"mov v16.16b, v17.16b",
"unimplemented (Unimplemented)",
"tbl v16.16b, {v16.16b}, v2.16b",
"mov x0, #0xff00000000",
"dup v1.2d, x0",
"eor v16.16b, v16.16b, v1.16b"
]
}
}
}
+91 -63
View File
@@ -16,294 +16,322 @@
"Instructions": {
"pi2fw mm0, mm1": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0x0f 0x0f 0x0c"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d2, [x28, #768]",
"uzp1 v2.4h, v2.4h, v2.4h",
"sxtl v2.4s, v2.4h",
"scvtf v2.2s, v2.2s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pi2fd mm0, mm1": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0x0f 0x0f 0x0d"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d2, [x28, #768]",
"scvtf v2.2s, v2.2s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pf2iw mm0, mm1": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0x0f 0x0f 0x1c"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d2, [x28, #768]",
"fcvtzs v2.2s, v2.2s",
"uzp1 v2.4h, v2.4h, v2.4h",
"sxtl v2.4s, v2.4h",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pf2id mm0, mm1": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0x0f 0x0f 0x1d"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d2, [x28, #768]",
"fcvtzs v2.2s, v2.2s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfrcpv mm0, mm1": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0x0f 0x0f 0x86"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d2, [x28, #768]",
"fmov v0.4s, #0x70 (1.0000)",
"fdiv v2.4s, v0.4s, v2.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfrsqrtv mm0, mm1": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0x0f 0x0f 0x87"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d2, [x28, #768]",
"fmov v0.4s, #0x70 (1.0000)",
"fsqrt v1.4s, v2.4s",
"fdiv v2.4s, v0.4s, v1.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfnacc mm0, mm1": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0x8a",
"ExpectedArm64ASM": [
"ldr d2, [x28, #768]",
"ldr d3, [x28, #784]",
"ldr d2, [x28, #752]",
"ldr d3, [x28, #768]",
"uzp1 v4.2s, v2.2s, v3.2s",
"uzp2 v2.2s, v2.2s, v3.2s",
"fsub v2.4s, v4.4s, v2.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfpnacc mm0, mm1": {
"ExpectedInstructionCount": 7,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0x8e",
"ExpectedArm64ASM": [
"ldr d2, [x28, #768]",
"ldr d3, [x28, #784]",
"ldr d2, [x28, #752]",
"ldr d3, [x28, #768]",
"dup v4.2s, v2.s[1]",
"fsub s2, s2, s4",
"faddp v3.4s, v3.4s, v3.4s",
"mov v2.s[1], v3.s[0]",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfcmpge mm0, mm1": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0x90",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"fcmge v2.4s, v3.4s, v2.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfmin mm0, mm1": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0x94",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"fcmgt v0.4s, v3.4s, v2.4s",
"bif v2.16b, v3.16b, v0.16b",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfrcp mm0, mm1": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0x0f 0x0f 0x96"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d2, [x28, #768]",
"fmov s0, #0x70 (1.0000)",
"fdiv s2, s0, s2",
"dup v2.2s, v2.s[0]",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfrsqrt mm0, mm1": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"0x0f 0x0f 0x97"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d2, [x28, #768]",
"fmov s0, #0x70 (1.0000)",
"fsqrt s1, s2",
"fdiv s2, s0, s1",
"dup v2.2s, v2.s[0]",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfsub mm0, mm1": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0x9a",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"fsub v2.4s, v3.4s, v2.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfadd mm0, mm1": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0x9e",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"fadd v2.4s, v3.4s, v2.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfcmpgt mm0, mm1": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xa0",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"fcmgt v2.4s, v3.4s, v2.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfmax mm0, mm1": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xa4",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"fcmgt v0.4s, v3.4s, v2.4s",
"bit v2.16b, v3.16b, v0.16b",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfrcpit1 mm0, mm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xa6",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"str d2, [x28, #768]"
"ldr d2, [x28, #768]",
"str d2, [x28, #752]"
]
},
"pfrcpit1 mm0, mm0": {
"ExpectedInstructionCount": 0,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xa6",
"ExpectedArm64ASM": []
},
"pfrsqit1 mm0, mm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xa7",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"str d2, [x28, #768]"
"ldr d2, [x28, #768]",
"str d2, [x28, #752]"
]
},
"pfrsqit1 mm0, mm0": {
"ExpectedInstructionCount": 0,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xa7",
"ExpectedArm64ASM": []
},
"pfsubr mm0, mm1": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xaa",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"fsub v2.4s, v2.4s, v3.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfcmpeq mm0, mm1": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xb0",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"fcmeq v2.4s, v3.4s, v2.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfmul mm0, mm1": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xb4",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"fmul v2.4s, v3.4s, v2.4s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pfrcpit2 mm0, mm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xb6",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"str d2, [x28, #768]"
"ldr d2, [x28, #768]",
"str d2, [x28, #752]"
]
},
"pfrcpit2 mm0, mm0": {
"ExpectedInstructionCount": 0,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xb6",
"ExpectedArm64ASM": []
},
"db 0x0f, 0x0f, 0xc1, 0xb7": {
"ExpectedInstructionCount": 7,
"Optimal": "Yes",
"Comment": [
"nasm doesn't support emitting this instruction",
"pmulhrw mm0, mm1",
"0x0f 0x0f 0xb7"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #768]",
"ldr d3, [x28, #784]",
"ldr d2, [x28, #752]",
"ldr d3, [x28, #768]",
"smull v2.4s, v2.4h, v3.4h",
"movi v3.4s, #0x80, lsl #8",
"add v2.4s, v2.4s, v3.4s",
"shrn v2.4h, v2.4s, #16",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pswapd mm0, mm1": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xbb",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d2, [x28, #768]",
"rev64 v2.2s, v2.2s",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
},
"pavgusb mm0, mm1": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": "0x0f 0x0f 0xbf",
"ExpectedArm64ASM": [
"ldr d2, [x28, #784]",
"ldr d3, [x28, #768]",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"urhadd v2.16b, v3.16b, v2.16b",
"str d2, [x28, #768]"
"str d2, [x28, #752]"
]
}
}
@@ -15,6 +15,7 @@
"Instructions": {
"push ax, bx": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Comment": [
"Mergable 16-bit pushes. May or may not be an optimization."
],
@@ -29,6 +30,7 @@
},
"push rax, rbx": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Comment": [
"Mergable 64-bit pushes"
],
@@ -43,6 +45,7 @@
},
"adds xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": [
"Redundant scalar adds that can get eliminated without AFP."
],
@@ -59,6 +62,7 @@
},
"positive movsb": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
@@ -78,6 +82,7 @@
},
"positive movsw": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
@@ -97,6 +102,7 @@
},
"positive movsd": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
@@ -116,6 +122,7 @@
},
"positive movsq": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
@@ -135,6 +142,7 @@
},
"negative movsb": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
@@ -154,6 +162,7 @@
},
"negative movsw": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
@@ -173,6 +182,7 @@
},
"negative movsd": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
@@ -192,6 +202,7 @@
},
"negative movsq": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
@@ -208,444 +219,6 @@
"sub x10, x10, #0x8 (8)",
"sub x11, x11, #0x8 (8)"
]
},
"positive rep movsb": {
"ExpectedInstructionCount": 18,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"cld",
"rep movsb"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
"cbz x0, #+0x14",
"ldrb w3, [x2], #1",
"strb w3, [x1], #1",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0xc",
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"add x22, x0, x2",
"add x23, x1, x2",
"mov x11, x22",
"mov x10, x23",
"mov x5, x20"
]
},
"positive rep movsw": {
"ExpectedInstructionCount": 18,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"cld",
"rep movsw"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
"cbz x0, #+0x14",
"ldrh w3, [x2], #2",
"strh w3, [x1], #2",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0xc",
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"add x22, x0, x2, lsl #1",
"add x23, x1, x2, lsl #1",
"mov x11, x22",
"mov x10, x23",
"mov x5, x20"
]
},
"positive rep movsd": {
"ExpectedInstructionCount": 18,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"cld",
"rep movsd"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
"cbz x0, #+0x14",
"ldr w3, [x2], #4",
"str w3, [x1], #4",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0xc",
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"add x22, x0, x2, lsl #2",
"add x23, x1, x2, lsl #2",
"mov x11, x22",
"mov x10, x23",
"mov x5, x20"
]
},
"positive rep movsq": {
"ExpectedInstructionCount": 18,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"cld",
"rep movsq"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
"cbz x0, #+0x14",
"ldr x3, [x2], #8",
"str x3, [x1], #8",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0xc",
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"add x22, x0, x2, lsl #3",
"add x23, x1, x2, lsl #3",
"mov x11, x22",
"mov x10, x23",
"mov x5, x20"
]
},
"negative rep movsb": {
"ExpectedInstructionCount": 18,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"std",
"rep movsb"
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
"cbz x0, #+0x14",
"ldrb w3, [x2], #-1",
"strb w3, [x1], #-1",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0xc",
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"sub x20, x0, x2",
"sub x21, x1, x2",
"mov x11, x20",
"mov x10, x21",
"mov w5, #0x0"
]
},
"negative rep movsw": {
"ExpectedInstructionCount": 18,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"std",
"rep movsw"
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
"cbz x0, #+0x14",
"ldrh w3, [x2], #-2",
"strh w3, [x1], #-2",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0xc",
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"sub x20, x0, x2, lsl #1",
"sub x21, x1, x2, lsl #1",
"mov x11, x20",
"mov x10, x21",
"mov w5, #0x0"
]
},
"negative rep movsd": {
"ExpectedInstructionCount": 18,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"std",
"rep movsd"
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
"cbz x0, #+0x14",
"ldr w3, [x2], #-4",
"str w3, [x1], #-4",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0xc",
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"sub x20, x0, x2, lsl #2",
"sub x21, x1, x2, lsl #2",
"mov x11, x20",
"mov x10, x21",
"mov w5, #0x0"
]
},
"negative rep movsq": {
"ExpectedInstructionCount": 18,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"std",
"rep movsq"
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
"cbz x0, #+0x14",
"ldr x3, [x2], #-8",
"str x3, [x1], #-8",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0xc",
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"sub x20, x0, x2, lsl #3",
"sub x21, x1, x2, lsl #3",
"mov x11, x20",
"mov x10, x21",
"mov w5, #0x0"
]
},
"positive rep stosb": {
"ExpectedInstructionCount": 11,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"cld",
"rep stosb"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"strb w20, [x28, #714]",
"uxtb w21, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x10",
"strb w21, [x1], #1",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"add x11, x11, x5",
"mov x5, x20"
]
},
"positive rep stosw": {
"ExpectedInstructionCount": 11,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"cld",
"rep stosw"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"strb w20, [x28, #714]",
"uxth w21, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x10",
"strh w21, [x1], #2",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"add x11, x11, x5, lsl #1",
"mov x5, x20"
]
},
"positive rep stosd": {
"ExpectedInstructionCount": 11,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"cld",
"rep stosd"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"strb w20, [x28, #714]",
"mov w21, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x10",
"str w21, [x1], #4",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"add x11, x11, x5, lsl #2",
"mov x5, x20"
]
},
"positive rep stosq": {
"ExpectedInstructionCount": 10,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"cld",
"rep stosq"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x10",
"str x4, [x1], #8",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"add x11, x11, x5, lsl #3",
"mov x5, x20"
]
},
"negative rep stosb": {
"ExpectedInstructionCount": 11,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"std",
"rep stosb"
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #714]",
"uxtb w20, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x10",
"strb w20, [x1], #-1",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"sub x11, x11, x5",
"mov w5, #0x0"
]
},
"negative rep stosw": {
"ExpectedInstructionCount": 11,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"std",
"rep stosw"
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #714]",
"uxth w20, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x10",
"strh w20, [x1], #-2",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"sub x11, x11, x5, lsl #1",
"mov w5, #0x0"
]
},
"negative rep stosd": {
"ExpectedInstructionCount": 11,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"std",
"rep stosd"
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #714]",
"mov w20, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x10",
"str w20, [x1], #-4",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"sub x11, x11, x5, lsl #2",
"mov w5, #0x0"
]
},
"negative rep stosq": {
"ExpectedInstructionCount": 10,
"Comment": [
"When direction flag is a compile time constant we can optimize",
"loads and stores can turn in to post-increment when known"
],
"x86Insts": [
"std",
"rep stosq"
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #714]",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x10",
"str x4, [x1], #-8",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"sub x11, x11, x5, lsl #3",
"mov w5, #0x0"
]
}
}
}
@@ -16,6 +16,7 @@
"Instructions": {
"adds xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Redundant scalar operations should get eliminated with AFP"
],
File diff suppressed because it is too large. Load diff
Loaded 100 of 145 files, more files were not shown because too many files have changed in this diff. Show more