mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-08 23:00:08 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
da069571f3 | ||
|
|
d2bac45b49 | ||
|
|
8913c59acc | ||
|
|
c3261b4aeb | ||
|
|
c7fb95aec5 | ||
|
|
c00cef6dc1 | ||
|
|
429ff94dc5 | ||
|
|
a6c67ca749 | ||
|
|
f51812a670 | ||
|
|
686294f1c4 | ||
|
|
a47ed105e7 | ||
|
|
b2d579a268 | ||
|
|
eb1050092f | ||
|
|
b3794f5541 | ||
|
|
1ecfa3253d | ||
|
|
5daf007b6a | ||
|
|
8efa5febd0 | ||
|
|
5fee8028cd | ||
|
|
6bc7a83c64 | ||
|
|
e55b5d0d11 | ||
|
|
19de7f2785 | ||
|
|
e32c5384ab | ||
|
|
b391fe6b92 | ||
|
|
4cfb81156f | ||
|
|
6121708e55 | ||
|
|
6ab214adea | ||
|
|
90b1ac4162 | ||
|
|
3abe6c14a1 | ||
|
|
fc1b500eff | ||
|
|
5d47b9195b | ||
|
|
a8272b74f6 | ||
|
|
12dc16780f | ||
|
|
b8af569841 | ||
|
|
8bee101795 | ||
|
|
2d66bc258a | ||
|
|
d2f86e49f7 | ||
|
|
d66cd16cfb | ||
|
|
04e785e434 | ||
|
|
15a1a0f7d9 | ||
|
|
0a58ce6134 | ||
|
|
8f5607f0e8 | ||
|
|
a21789d3d8 | ||
|
|
efd6e95059 | ||
|
|
9bdb1f4306 | ||
|
|
ae4b7135d5 | ||
|
|
d503366816 | ||
|
|
0fe2827fcc | ||
|
|
cd6722f77b | ||
|
|
ffb745b662 | ||
|
|
bb10f25808 | ||
|
|
aa1076d12b | ||
|
|
1e827ec7a6 | ||
|
|
3fe2650787 | ||
|
|
3e99e814bc | ||
|
|
2019f8138e | ||
|
|
e44d1f136b | ||
|
|
09872402df | ||
|
|
b078a41a02 | ||
|
|
3a5eeb5700 | ||
|
|
9433ae3405 | ||
|
|
4658b24f9a | ||
|
|
4e7d0e6be0 | ||
|
|
4ddd98708f | ||
|
|
c161fd218c | ||
|
|
7e257cc268 | ||
|
|
d8ef70280c | ||
|
|
ec003281be | ||
|
|
e58f67b76c | ||
|
|
57178abcd2 | ||
|
|
73ca4f8314 | ||
|
|
527752c25b | ||
|
|
38fa866c91 | ||
|
|
c902b8807a | ||
|
|
4934c1fd94 | ||
|
|
766fbe3db3 | ||
|
|
29405f2690 | ||
|
|
51f505acca | ||
|
|
77415538f7 | ||
|
|
9fb69ed206 | ||
|
|
735a4f90db | ||
|
|
7ef8dc13ba | ||
|
|
656477ec63 | ||
|
|
d080180e85 | ||
|
|
af1d2d6005 | ||
|
|
90c1282f3a | ||
|
|
d5d7eec8b0 | ||
|
|
5337b9537d | ||
|
|
e72c016230 | ||
|
|
072cf4c5bd | ||
|
|
27ededf47f | ||
|
|
82d7f9fdd7 | ||
|
|
d85153d6b3 | ||
|
|
6b698e6cd1 | ||
|
|
9475f79ec6 | ||
|
|
f906c6a0f4 | ||
|
|
e88c92de57 | ||
|
|
48ed906a7b | ||
|
|
b03b02d2f2 | ||
|
|
d00d476a0a | ||
|
|
ac1e32994a | ||
|
|
800d447f3d | ||
|
|
8111b7cc7f | ||
|
|
a86c922073 | ||
|
|
46fb8583bb | ||
|
|
3487d120ec | ||
|
|
07394d6a6e | ||
|
|
8d3204171c | ||
|
|
7641f722e9 | ||
|
|
b51fa497c5 | ||
|
|
34722bed3d | ||
|
|
981c3009ee | ||
|
|
38cf357d85 | ||
|
|
2533ed4a63 | ||
|
|
5a4691fdfc | ||
|
|
6c035a0d61 | ||
|
|
a234aa300d | ||
|
|
f6abbedbd1 | ||
|
|
b6fe4cd6dd | ||
|
|
bdae4f6915 | ||
|
|
f8b6edfb2b | ||
|
|
0a1ecdf6ae | ||
|
|
beec203f56 | ||
|
|
d323032ec9 | ||
|
|
7472b21f33 | ||
|
|
1058575d3a | ||
|
|
e9867ca35a | ||
|
|
84277319fa | ||
|
|
8f8aa55c7f | ||
|
|
72a4063651 | ||
|
|
0b1229da55 | ||
|
|
1d3ce30e50 | ||
|
|
71187d3ad7 | ||
|
|
dd8a3a9aea | ||
|
|
7b2fc37651 | ||
|
|
426569d74d | ||
|
|
e877d5b82c | ||
|
|
572e0d04d5 | ||
|
|
e3d7161ac5 | ||
|
|
fcbf0de05a |
No files matched your search
@@ -37,6 +37,7 @@ option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
|
||||
@@ -479,6 +480,7 @@ if (BUILD_THUNKS)
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
@@ -497,6 +499,7 @@ if (BUILD_THUNKS)
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
|
||||
Vendored
+1
-1
Submodule External/fmt updated: 0c9fce2ffe...873670ba3f.
@@ -44,7 +44,7 @@ class OpDefinition:
|
||||
HasDest: bool
|
||||
DestType: str
|
||||
DestSize: str
|
||||
NumElements: str
|
||||
ElementSize: str
|
||||
OpClass: str
|
||||
HasSideEffects: bool
|
||||
ImplicitFlagClobber: bool
|
||||
@@ -67,7 +67,7 @@ class OpDefinition:
|
||||
self.HasDest = False
|
||||
self.DestType = None
|
||||
self.DestSize = None
|
||||
self.NumElements = None
|
||||
self.ElementSize = None
|
||||
self.OpClass = None
|
||||
self.OpSize = 0
|
||||
self.HasSideEffects = False
|
||||
@@ -101,7 +101,8 @@ def is_ssa_type(type):
|
||||
if (type == "SSA" or
|
||||
type == "GPR" or
|
||||
type == "GPRPair" or
|
||||
type == "FPR"):
|
||||
type == "FPR" or
|
||||
type == "PRED"):
|
||||
return True
|
||||
return False
|
||||
|
||||
@@ -150,8 +151,8 @@ def parse_ops(ops):
|
||||
RHS += f", {DType}:$Out{Name}"
|
||||
else:
|
||||
# Single anonymous destination
|
||||
if LHS not in ["SSA", "GPR", "GPRPair", "FPR"]:
|
||||
ExitError(f"Unknown destination class type {LHS}. Needs to be one of SSA, GPR, GPRPair, FPR")
|
||||
if LHS not in ["SSA", "GPR", "GPRPair", "FPR", "PRED"]:
|
||||
ExitError(f"Unknown destination class type {LHS}. Needs to be one of SSA, GPR, GPRPair, FPR, PRED")
|
||||
|
||||
OpDef.HasDest = True
|
||||
OpDef.DestType = LHS
|
||||
@@ -221,7 +222,8 @@ def parse_ops(ops):
|
||||
if (OpArg.IsSSA and
|
||||
(OpArg.Type == "GPR" or
|
||||
OpArg.Type == "GPRPair" or
|
||||
OpArg.Type == "FPR")):
|
||||
OpArg.Type == "FPR" or
|
||||
OpArg.Type == "PRED")):
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
|
||||
|
||||
OpArg.Name = ArgName
|
||||
@@ -232,8 +234,8 @@ def parse_ops(ops):
|
||||
if "DestSize" in op_val:
|
||||
OpDef.DestSize = op_val["DestSize"]
|
||||
|
||||
if "NumElements" in op_val:
|
||||
OpDef.NumElements = op_val["NumElements"]
|
||||
if "ElementSize" in op_val:
|
||||
OpDef.ElementSize = op_val["ElementSize"]
|
||||
|
||||
if len(op_class):
|
||||
OpDef.OpClass = op_class
|
||||
@@ -743,10 +745,10 @@ def print_ir_allocator_helpers():
|
||||
if op.DestSize != None:
|
||||
output_file.write("\t\t_Op.first->Header.Size = {};\n".format(op.DestSize))
|
||||
|
||||
if op.NumElements == None:
|
||||
if op.ElementSize == None:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size;\n")
|
||||
else:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size / ({});\n".format(op.NumElements))
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = {};\n".format(op.ElementSize))
|
||||
|
||||
# Insert validation here
|
||||
if op.EmitValidation != None:
|
||||
|
||||
@@ -88,8 +88,11 @@ public:
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState* Thread, uint64_t RIP) override;
|
||||
|
||||
bool IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) override;
|
||||
bool IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) override;
|
||||
|
||||
void ReconstructXMMRegisters(const FEXCore::Core::InternalThreadState* Thread, __uint128_t* XMM_Low, __uint128_t* YMM_High) override;
|
||||
@@ -288,6 +291,7 @@ public:
|
||||
[[nodiscard]]
|
||||
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
|
||||
|
||||
IR::OpSize GetGPROpSize() const {
|
||||
return Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
|
||||
|
||||
@@ -57,6 +57,12 @@ namespace x64 {
|
||||
ARMEmitter::Reg::r24, ARMEmitter::Reg::r25, ARMEmitter::Reg::r30, ARMEmitter::Reg::r18,
|
||||
};
|
||||
|
||||
// p6 and p7 registers are used as temporaries no not added here for RA
|
||||
// See PREF_TMP_16B and PREF_TMP_32B
|
||||
// p0-p1 are also used in the jit as temps.
|
||||
// Also p8-p15 cannot be used can only encode p0-p7, so we're left with p2-p5.
|
||||
constexpr std::array<ARMEmitter::PRegister, 4> PR = {ARMEmitter::PReg::p2, ARMEmitter::PReg::p3, ARMEmitter::PReg::p4, ARMEmitter::PReg::p5};
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
// All are caller saved
|
||||
@@ -103,6 +109,12 @@ namespace x64 {
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
// p6 and p7 registers are used as temporaries no not added here for RA
|
||||
// See PREF_TMP_16B and PREF_TMP_32B
|
||||
// p0-p1 are also used in the jit as temps.
|
||||
// Also p8-p15 cannot be used can only encode p0-p7, so we're left with p2-p5.
|
||||
constexpr std::array<ARMEmitter::PRegister, 4> PR = {ARMEmitter::PReg::p2, ARMEmitter::PReg::p3, ARMEmitter::PReg::p4, ARMEmitter::PReg::p5};
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
@@ -234,6 +246,12 @@ namespace x32 {
|
||||
|
||||
constexpr unsigned RAPairs = 12;
|
||||
|
||||
// p6 and p7 registers are used as temporaries no not added here for RA
|
||||
// See PREF_TMP_16B and PREF_TMP_32B
|
||||
// p0-p1 are also used in the jit as temps.
|
||||
// Also p8-p15 cannot be used can only encode p0-p7, so we're left with p2-p5.
|
||||
constexpr std::array<ARMEmitter::PRegister, 4> PR = {ARMEmitter::PReg::p2, ARMEmitter::PReg::p3, ARMEmitter::PReg::p4, ARMEmitter::PReg::p5};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
ARMEmitter::VReg::v16, ARMEmitter::VReg::v17, ARMEmitter::VReg::v18, ARMEmitter::VReg::v19,
|
||||
@@ -357,6 +375,7 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
GeneralRegisters = x64::RA;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
PredicateRegisters = x64::PR;
|
||||
PairRegisters = x64::RAPairs;
|
||||
#ifdef _M_ARM_64EC
|
||||
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
|
||||
@@ -370,6 +389,8 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
|
||||
PredicateRegisters = x32::PR;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -76,6 +76,9 @@ constexpr size_t CPU_AREA_EMULATOR_STACK_BASE_OFFSET = 0x8;
|
||||
constexpr size_t CPU_AREA_EMULATOR_DATA_OFFSET = 0x30;
|
||||
#endif
|
||||
|
||||
// Will force one single instruction block to be generated first if set when entering the JIT filling SRA.
|
||||
constexpr auto ENTRY_FILL_SRA_SINGLE_INST_REG = TMP1;
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
|
||||
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
|
||||
@@ -94,6 +97,7 @@ protected:
|
||||
std::span<const ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegisters {};
|
||||
std::span<const ARMEmitter::PRegister> PredicateRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> StaticFPRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
|
||||
@@ -39,6 +39,15 @@ namespace CPU {
|
||||
{0xC90F'DAA2'2168'C235ULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_PI
|
||||
{0x9A20'9A84'FBCF'F799ULL, 0x0000'0000'0000'3FFDULL}, // NAMED_VECTOR_X87_LOG10_2
|
||||
{0xB172'17F7'D1CF'79ACULL, 0x0000'0000'0000'3FFEULL}, // NAMED_VECTOR_X87_LOG_2
|
||||
{0x4F00'0000'4F00'0000ULL, 0x4F00'0000'4F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I32
|
||||
{0x4F00'0000'4F00'0000ULL, 0x4F00'0000'4F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I32_UPPER
|
||||
{0x5F00'0000'5F00'0000ULL, 0x5F00'0000'5F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I64
|
||||
{0x41E0'0000'0000'0000ULL, 0x41E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I32
|
||||
{0x41E0'0000'0000'0000ULL, 0x41E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I32_UPPER
|
||||
{0x43E0'0000'0000'0000ULL, 0x43E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I64
|
||||
{0x8000'0000'8000'0000ULL, 0x8000'0000'8000'0000ULL}, // NAMED_VECTOR_CVTMAX_I32
|
||||
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_I64
|
||||
{0x0000'0000'0000'0000ULL, 0x0000'0000'0000'8000ULL}, // NAMED_VECTOR_F80_SIGN_MASK
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {[]() consteval {
|
||||
|
||||
@@ -80,9 +80,16 @@ namespace CPU {
|
||||
struct JITCodeTail {
|
||||
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
|
||||
size_t Size;
|
||||
|
||||
// RIP that the block's entry comes from.
|
||||
uint64_t RIP;
|
||||
|
||||
// The length of the guest code for this block.
|
||||
size_t GuestSize;
|
||||
|
||||
// If this block represents a single guest instruction.
|
||||
bool SingleInst;
|
||||
|
||||
// Number of RIP entries for this JIT Code section.
|
||||
uint32_t NumberOfRIPEntries;
|
||||
|
||||
@@ -119,14 +126,17 @@ namespace CPU {
|
||||
*
|
||||
* This is a thread specific compilation unit since there is one CPUBackend per guest thread
|
||||
*
|
||||
* @param Size - The byte size of the guest code for this block
|
||||
* @param SingleInst - If this block represents a single guest instruction
|
||||
* @param IR - IR that maps to the IR for this RIP
|
||||
* @param DebugData - Debug data that is available for this IR indirectly
|
||||
* @param CheckTF - If EFLAGS.TF checks should be emitted at the start of the block
|
||||
*
|
||||
* @return Information about the compiled code block.
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) = 0;
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
|
||||
@@ -112,13 +112,38 @@ ContextImpl::~ContextImpl() {
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
struct GetFrameBlockInfoResult {
|
||||
const CPU::CPUBackend::JITCodeHeader* InlineHeader;
|
||||
const CPU::CPUBackend::JITCodeTail* InlineTail;
|
||||
};
|
||||
static GetFrameBlockInfoResult GetFrameBlockInfo(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader*>(BlockBegin);
|
||||
|
||||
if (InlineHeader) {
|
||||
auto InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail*>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
return {InlineHeader, InlineTail};
|
||||
}
|
||||
|
||||
return {InlineHeader, nullptr};
|
||||
}
|
||||
|
||||
bool ContextImpl::IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) {
|
||||
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
return InlineTail && (Address + Size > InlineTail->RIP && Address < InlineTail->RIP + InlineTail->GuestSize);
|
||||
}
|
||||
|
||||
bool ContextImpl::IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) {
|
||||
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
return InlineTail && InlineTail->SingleInst;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
auto [InlineHeader, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
|
||||
if (InlineHeader) {
|
||||
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries*>(
|
||||
Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
@@ -150,7 +175,8 @@ uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* T
|
||||
return Frame->State.rip;
|
||||
}
|
||||
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) {
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs,
|
||||
uint64_t PSTATE) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
uint32_t EFLAGS {};
|
||||
|
||||
@@ -160,6 +186,7 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
case X86State::RFLAG_CF_RAW_LOC:
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
case X86State::RFLAG_AF_RAW_LOC:
|
||||
case X86State::RFLAG_TF_RAW_LOC:
|
||||
case X86State::RFLAG_ZF_RAW_LOC:
|
||||
case X86State::RFLAG_SF_RAW_LOC:
|
||||
case X86State::RFLAG_OF_RAW_LOC:
|
||||
@@ -212,6 +239,9 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
uint32_t AF = ((Frame->State.af_raw ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
|
||||
|
||||
uint8_t TFByte = Frame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
EFLAGS |= (TFByte & 1) << X86State::RFLAG_TF_RAW_LOC;
|
||||
|
||||
// DF is pretransformed, undo the transform from 1/-1 back to 0/1
|
||||
uint8_t DFByte = Frame->State.flags[X86State::RFLAG_DF_RAW_LOC];
|
||||
if (DFByte & 0x80) {
|
||||
@@ -366,7 +396,7 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
|
||||
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
@@ -469,7 +499,7 @@ void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, ui
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
@@ -550,6 +580,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
|
||||
|
||||
bool HadDispatchError {false};
|
||||
bool HadInvalidInst {false};
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, MaxInst,
|
||||
[Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
@@ -647,16 +678,23 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
++TotalInstructions;
|
||||
}
|
||||
} else {
|
||||
if (TableInfo) {
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
if (!BlockInstructionsLength) {
|
||||
// SMC can modify block contents and patch invalid instructions to valid ones inline.
|
||||
// End blocks upon encountering them and only emit an invalid opcode exception if there are no prior instructions in the block (that could have modified it to be valid).
|
||||
|
||||
if (TableInfo) {
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
}
|
||||
|
||||
HadInvalidInst = true;
|
||||
}
|
||||
|
||||
const bool NeedsBlockEnd =
|
||||
(HadDispatchError && TotalInstructions > 0) || (Thread->OpDispatcher->NeedsBlockEnder() && i + 1 == InstsInBlock);
|
||||
const bool NeedsBlockEnd = (HadDispatchError && TotalInstructions > 0) ||
|
||||
(Thread->OpDispatcher->NeedsBlockEnder() && i + 1 == InstsInBlock) || HadInvalidInst;
|
||||
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
@@ -742,6 +780,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
uint64_t TotalInstructions {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
|
||||
@@ -759,11 +798,12 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
|
||||
if (!IR) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
auto [IRCopy, _TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IR = std::move(IRCopy);
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
TotalInstructions = _TotalInstructions;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
}
|
||||
@@ -771,13 +811,17 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
if (!IR) {
|
||||
return {};
|
||||
}
|
||||
|
||||
// If the trap flag is set we generate single instruction blocks that each check to generate a single step exception.
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
auto IRView = IR->GetIRView();
|
||||
return {
|
||||
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
|
||||
// In the future with code caching getting wired up, we will pass the rest of the data forward.
|
||||
// TODO: Pass the data forward when code caching is wired up to this.
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, &IRView, DebugData, IR->RAData()).BlockEntry,
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &IRView, DebugData, IR->RAData(), TFSet).BlockEntry,
|
||||
.IR = std::move(IR),
|
||||
.DebugData = DebugData,
|
||||
.GeneratedIR = true,
|
||||
@@ -862,6 +906,24 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileSingleStep");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
auto [CodePtr, IR, DebugData, GeneratedIR, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
|
||||
@@ -46,6 +46,8 @@ Dispatcher::~Dispatcher() {
|
||||
}
|
||||
|
||||
void Dispatcher::EmitDispatcher() {
|
||||
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
|
||||
auto RipReg = TMP3;
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
#endif
|
||||
@@ -62,7 +64,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ARMEmitter::ForwardLabel l_CTX;
|
||||
ARMEmitter::SingleUseForwardLabel l_Sleep;
|
||||
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
|
||||
ARMEmitter::ForwardLabel l_CompileBlock;
|
||||
ARMEmitter::ForwardLabel l_CompileSingleStep;
|
||||
|
||||
// Push all the register we need to save
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -81,6 +84,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
FillStaticRegs();
|
||||
ARMEmitter::BiDirectionalLabel LoopTop {};
|
||||
ARMEmitter::ForwardLabel CompileSingleStep;
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
b(&LoopTop);
|
||||
@@ -89,6 +93,10 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
// Force a single instruction block if ENTRY_FILL_SRA_SINGLE_INST_REG is nonzero entering the JIT, used for inline SMC handling.
|
||||
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
|
||||
// Enter JIT
|
||||
b(&LoopTop);
|
||||
|
||||
@@ -116,10 +124,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
|
||||
auto RipReg = TMP3;
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
@@ -204,37 +213,21 @@ void Dispatcher::EmitDispatcher() {
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// Clobbers TMP1/2
|
||||
auto EmitSignalGuardedRegion = [&](auto Body) {
|
||||
#ifndef _WIN32
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
|
||||
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
Body();
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
@@ -250,15 +243,36 @@ void Dispatcher::EmitDispatcher() {
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
};
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
});
|
||||
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Clobbers TMP1/2
|
||||
auto EmitECExitCheck = [&]() {
|
||||
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
|
||||
ARMEmitter::SingleUseForwardLabel l_NotECCode;
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
@@ -277,56 +291,83 @@ void Dispatcher::EmitDispatcher() {
|
||||
br(TMP2);
|
||||
|
||||
Bind(&l_NotECCode);
|
||||
};
|
||||
#endif
|
||||
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
#endif
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
mov(ARMEmitter::XReg::x3, 0);
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
|
||||
}
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
mov(ARMEmitter::XReg::x3, 0);
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
|
||||
}
|
||||
|
||||
// Result is now in x0
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
});
|
||||
|
||||
// Jump to the compiled block
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
{
|
||||
Bind(&CompileSingleStep);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
strb(ARMEmitter::WReg::zr, TMP1, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x2, RipReg);
|
||||
}
|
||||
|
||||
b(&LoopTop);
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileSingleStep);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP }
|
||||
}
|
||||
|
||||
// Result is now in x0
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
});
|
||||
|
||||
// Jump to the compiled block
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -505,8 +546,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
Bind(&l_Sleep);
|
||||
dc64(reinterpret_cast<uint64_t>(SleepThread));
|
||||
Bind(&l_CompileBlock);
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMF.GetConvertedPointer());
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMFCompileBlock.GetConvertedPointer());
|
||||
Bind(&l_CompileSingleStep);
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
|
||||
dc64(PMFCompileSingleStep.GetConvertedPointer());
|
||||
|
||||
Start = reinterpret_cast<uint64_t>(DispatchPtr);
|
||||
End = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -220,7 +220,8 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
{
|
||||
// If we have a VSIB byte (as opposed to SIB), then the index register is a vector.
|
||||
const bool IsIndexVector = (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
// DecodeInst->TableInfo may be null in the case of 3DNow! ModRM decoding.
|
||||
const bool IsIndexVector = DecodeInst->TableInfo && (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
uint8_t InvalidSIBIndex = 0b100; ///< SIB Index where there is no register encoding.
|
||||
if (IsIndexVector) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_VSIB_BYTE;
|
||||
|
||||
@@ -6,17 +6,20 @@
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
|
||||
softfloat_state State {};
|
||||
State.detectTininess = softfloat_tininess_afterRounding;
|
||||
State.exceptionFlags = 0;
|
||||
State.roundingPrecision = 80;
|
||||
|
||||
auto PC = (FCW >> 8) & 3;
|
||||
switch (PC) {
|
||||
case 0: State.roundingPrecision = 32; break;
|
||||
case 2: State.roundingPrecision = 64; break;
|
||||
case 3: State.roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
if (!Force80BitPrecision) {
|
||||
auto PC = (FCW >> 8) & 3;
|
||||
switch (PC) {
|
||||
case 0: State.roundingPrecision = 32; break;
|
||||
case 2: State.roundingPrecision = 64; break;
|
||||
case 3: State.roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
}
|
||||
}
|
||||
|
||||
auto RC = (FCW >> 10) & 3;
|
||||
@@ -132,7 +135,7 @@ struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FRNDINT(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -140,7 +143,7 @@ struct OpHandlers<IR::OP_F80ROUND> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::F2XM1(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -148,7 +151,7 @@ struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FTAN(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -164,7 +167,7 @@ struct OpHandlers<IR::OP_F80SQRT> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSIN(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -172,7 +175,7 @@ struct OpHandlers<IR::OP_F80SIN> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FCOS(&State, Src1);
|
||||
}
|
||||
};
|
||||
@@ -226,7 +229,7 @@ struct OpHandlers<IR::OP_F80DIV> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FYL2X(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
@@ -234,7 +237,7 @@ struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FATAN(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
@@ -242,7 +245,7 @@ struct OpHandlers<IR::OP_F80ATAN> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM1(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
@@ -250,7 +253,7 @@ struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
@@ -258,7 +261,7 @@ struct OpHandlers<IR::OP_F80FPREM> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSCALE(&State, Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -484,11 +484,16 @@ static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
uintptr_t HostCode {};
|
||||
auto GuestRip = Record->GuestRIP;
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
if (!TFSet) {
|
||||
HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
}
|
||||
|
||||
if (!HostCode) {
|
||||
if (TFSet || !HostCode) {
|
||||
// If TF is set, the cache must be skipped as different code needs to be generated.
|
||||
Frame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
@@ -534,6 +539,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::PREDClass, PredicateRegisters.size());
|
||||
RAPass->PairRegs = PairRegisters;
|
||||
|
||||
{
|
||||
@@ -654,8 +660,69 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
if (CheckTF) {
|
||||
ARMEmitter::SingleUseForwardLabel l_TFUnset;
|
||||
ARMEmitter::SingleUseForwardLabel l_TFBlocked;
|
||||
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
tbz(TMP1, 1, &l_TFBlocked);
|
||||
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Core::FAULT_SIGTRAP,
|
||||
.TrapNo = X86State::X86_TRAPNO_DB,
|
||||
.si_code = 2,
|
||||
.err_code = 0,
|
||||
};
|
||||
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
|
||||
Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
Bind(&l_TFUnset);
|
||||
}
|
||||
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::SingleUseForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
#endif
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData,
|
||||
bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
@@ -711,22 +778,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
adr(TMP1, &JITCodeHeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::SingleUseForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
#endif
|
||||
EmitInterruptChecks(CheckTF);
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
@@ -810,6 +862,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
JITBlockTail->GuestSize = Size;
|
||||
JITBlockTail->SingleInst = SingleInst;
|
||||
JITBlockTail->SpinLockFutex = 0;
|
||||
|
||||
{
|
||||
|
||||
@@ -38,8 +38,9 @@ public:
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) override;
|
||||
CPUBackend::CompiledCode
|
||||
CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) override;
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
@@ -94,6 +95,19 @@ private:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::PRegister GetPReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::PREDClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::PREDClass.Val) {
|
||||
return PredicateRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
@@ -333,6 +347,9 @@ private:
|
||||
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
|
||||
void EmitInterruptChecks(bool CheckTF);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
|
||||
@@ -10,6 +10,7 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
@@ -1551,6 +1552,75 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(InitPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_InitPredicate>();
|
||||
const auto OpSize = IROp->Size;
|
||||
ptrue(ConvertSubRegSize16(OpSize), GetPReg(Node), static_cast<ARMEmitter::PredicatePattern>(Op->Pattern));
|
||||
}
|
||||
|
||||
DEF_OP(StoreMemPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemPredicate>();
|
||||
const auto Predicate = GetPReg(Op->Mask.ID());
|
||||
|
||||
const auto RegData = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "StoreMemPredicate needs SVE support");
|
||||
|
||||
const auto MemDst = ARMEmitter::SVEMemOperand(MemReg.X(), 0);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(RegData.Z(), Predicate, MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i16Bit: {
|
||||
st1h<ARMEmitter::SubRegSize::i16Bit>(RegData.Z(), Predicate, MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
st1w<ARMEmitter::SubRegSize::i32Bit>(RegData.Z(), Predicate, MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
st1d(RegData.Z(), Predicate, MemDst);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} element size: {}", __func__, IROp->ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMemPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPredicate>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Predicate = GetPReg(Op->Mask.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "LoadMemPredicate needs SVE support");
|
||||
|
||||
const auto MemDst = ARMEmitter::SVEMemOperand(MemReg.X(), 0);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), Predicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i16Bit: {
|
||||
ld1h<ARMEmitter::SubRegSize::i16Bit>(Dst.Z(), Predicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
ld1w<ARMEmitter::SubRegSize::i32Bit>(Dst.Z(), Predicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ld1d(Dst.Z(), Predicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} element size: {}", __func__, IROp->ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -267,7 +267,7 @@ DEF_OP(RDRAND) {
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
wfe();
|
||||
yield();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
@@ -6,6 +6,7 @@ desc: Handles x86/64 ops to IR, no-pf opt, local-flags opt
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/HostFeatures.h"
|
||||
#include "FEXCore/Utils/Telemetry.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
@@ -444,7 +445,7 @@ void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_idx));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
} else {
|
||||
switch (SegmentReg) {
|
||||
@@ -466,7 +467,7 @@ void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -517,6 +518,8 @@ void OpDispatchBuilder::POPSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
|
||||
// Unset the 'active' bit in the packed TF, skipping the single step exception after this instruction
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_TF_RAW_LOC>(_And(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::RFLAG_TF_RAW_LOC), _Constant(1)));
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
|
||||
@@ -3610,16 +3613,16 @@ void OpDispatchBuilder::DIVOp(OpcodeArgs) {
|
||||
auto ResultAX = _Bfi(GPRSize, 8, 8, UDivOp, URemOp);
|
||||
StoreGPRRegister(X86State::REG_RAX, ResultAX, OpSize::i16Bit);
|
||||
} else if (Size == OpSize::i16Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX, Size);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
auto UDivOp = _LUDiv(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
auto URemOp = _LURem(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp, Size);
|
||||
} else if (Size == OpSize::i32Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX, Size);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
Ref UDivOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LUDiv(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
Ref URemOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LURem(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
@@ -3651,7 +3654,7 @@ void OpDispatchBuilder::IDIVOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
if (Size == OpSize::i8Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, OpSize::i16Bit);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Src1 = _Sbfe(OpSize::i64Bit, 16, 0, Src1);
|
||||
Divisor = _Sbfe(OpSize::i64Bit, 8, 0, Divisor);
|
||||
|
||||
@@ -3662,16 +3665,16 @@ void OpDispatchBuilder::IDIVOp(OpcodeArgs) {
|
||||
auto ResultAX = _Bfi(GPRSize, 8, 8, UDivOp, URemOp);
|
||||
StoreGPRRegister(X86State::REG_RAX, ResultAX, OpSize::i16Bit);
|
||||
} else if (Size == OpSize::i16Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX, Size);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
auto UDivOp = _LDiv(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
auto URemOp = _LRem(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp, Size);
|
||||
} else if (Size == OpSize::i32Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX, Size);
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
Ref UDivOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LDiv(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
Ref URemOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LRem(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
@@ -4309,10 +4312,15 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
if ((IsOperandMem(Operand, true) && LoadData) || ForceLoad) {
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemSrc = LoadEffectiveAddress(A, true);
|
||||
|
||||
// For X87 extended doubles, Split the load.
|
||||
auto Res = _LoadMem(Class, OpSize::i64Bit, MemSrc, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
return _VLoadVectorElement(OpSize::i128Bit, OpSize::i16Bit, Res, 4, _Add(OpSize::i64Bit, MemSrc, _InlineConstant(8)));
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
// Using SVE we can load this with a single instruction.
|
||||
auto PReg = InitPredicateCached(OpSize::i16Bit, ARMEmitter::PredicatePattern::SVE_VL5);
|
||||
return _LoadMemPredicate(OpSize::i128Bit, OpSize::i16Bit, PReg, MemSrc);
|
||||
} else {
|
||||
// For X87 extended doubles, Split the load.
|
||||
auto Res = _LoadMem(Class, OpSize::i64Bit, MemSrc, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
return _VLoadVectorElement(OpSize::i128Bit, OpSize::i16Bit, Res, 4, _Add(OpSize::i64Bit, MemSrc, _InlineConstant(8)));
|
||||
}
|
||||
}
|
||||
|
||||
return _LoadMemAutoTSO(Class, OpSize, A, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
@@ -4439,11 +4447,15 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemStoreDst = LoadEffectiveAddress(A, true);
|
||||
|
||||
// For X87 extended doubles, split before storing
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, MemStoreDst, Src, Align);
|
||||
auto Upper = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, Src, 1);
|
||||
_StoreMem(GPRClass, OpSize::i16Bit, Upper, MemStoreDst, _Constant(8), std::min(Align, OpSize::i64Bit), MEM_OFFSET_SXTX, 1);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
auto PReg = InitPredicateCached(OpSize::i16Bit, ARMEmitter::PredicatePattern::SVE_VL5);
|
||||
_StoreMemPredicate(OpSize::i128Bit, OpSize::i16Bit, Src, PReg, MemStoreDst);
|
||||
} else {
|
||||
// For X87 extended doubles, split before storing
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, MemStoreDst, Src, Align);
|
||||
auto Upper = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, Src, 1);
|
||||
_StoreMem(GPRClass, OpSize::i16Bit, Upper, MemStoreDst, _Constant(8), std::min(Align, OpSize::i64Bit), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
_StoreMemAutoTSO(Class, OpSize, A, Src, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
}
|
||||
@@ -4951,9 +4963,11 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
#define PF_3A_66 1
|
||||
constexpr static std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F3A_AES[] = {
|
||||
{OPD(0, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
{OPD(1, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
};
|
||||
constexpr static std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F3A_PCLMUL[] = {
|
||||
{OPD(0, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
|
||||
{OPD(1, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
|
||||
};
|
||||
|
||||
#undef PF_3A_NONE
|
||||
@@ -5077,9 +5091,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{OPD(1, 0b00, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
@@ -5179,9 +5193,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b01, 0xE4), 1, &OpDispatchBuilder::VPMULHWOp<false>},
|
||||
{OPD(1, 0b01, 0xE5), 1, &OpDispatchBuilder::VPMULHWOp<true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::AVXVector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE7), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
|
||||
|
||||
@@ -125,6 +125,9 @@ public:
|
||||
|
||||
// Need to clear any named constants that were cached.
|
||||
ClearCachedNamedConstants();
|
||||
|
||||
// Clear predicate cache for x87 ldst
|
||||
ResetInitPredicateCache();
|
||||
}
|
||||
|
||||
IRPair<IROp_Jump> Jump() {
|
||||
@@ -466,10 +469,10 @@ public:
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void Scalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
void Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize, bool IsAVX);
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
void MASKMOVOp(OpcodeArgs);
|
||||
void MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType);
|
||||
@@ -515,12 +518,6 @@ public:
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVXScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void AVXVector_CVT_Float_To_Int(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Widen>
|
||||
void AVXVector_CVT_Int_To_Float(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void VectorScalarInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
@@ -1029,7 +1026,7 @@ public:
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVX128_Vector_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void AVX128_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Widen>
|
||||
@@ -1468,7 +1465,10 @@ private:
|
||||
Ref Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
Ref Vector_CVT_Float_To_IntImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode);
|
||||
Ref CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElementSize, bool HostRoundingMode);
|
||||
|
||||
Ref Vector_CVT_Float_To_Int32Impl(OpcodeArgs, IR::OpSize DstSize, Ref Src, IR::OpSize SrcSize, IR::OpSize SrcElementSize,
|
||||
bool HostRoundingMode, bool ZeroUpperHalf);
|
||||
|
||||
Ref Vector_CVT_Int_To_FloatImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen);
|
||||
|
||||
@@ -1788,6 +1788,13 @@ private:
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
|
||||
// For DF, we need to transform 0/1 into 1/-1
|
||||
StoreDF(_SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
|
||||
auto PackedTF = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
|
||||
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
|
||||
// unblocked bit before raising the exception.
|
||||
auto NewPackedTF = _Select(FEXCore::IR::COND_EQ, Value, _Constant(0), _And(OpSize::i32Bit, PackedTF, _Constant(~1)), _Constant(1));
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
} else {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
@@ -1941,6 +1948,7 @@ private:
|
||||
|
||||
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
|
||||
LOGMAN_THROW_AA_FMT(Index != DFIndex, "must be pairable");
|
||||
LOGMAN_THROW_AA_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
|
||||
// Try to load a pair into the cache
|
||||
uint64_t Bits = (3ull << (uint64_t)Index);
|
||||
@@ -2420,6 +2428,7 @@ private:
|
||||
}
|
||||
|
||||
AddressMode SelectPairAddressMode(AddressMode A, IR::OpSize Size) {
|
||||
LOGMAN_THROW_AA_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
const auto SizeInt = IR::OpSizeToSize(Size);
|
||||
AddressMode Out {};
|
||||
|
||||
|
||||
@@ -116,8 +116,8 @@ void OpDispatchBuilder::InstallAVX128Handlers() {
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{OPD(1, 0b00, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
@@ -217,9 +217,9 @@ void OpDispatchBuilder::InstallAVX128Handlers() {
|
||||
{OPD(1, 0b01, 0xE4), 1, &OpDispatchBuilder::AVX128_VPMULHW<false>},
|
||||
{OPD(1, 0b01, 0xE5), 1, &OpDispatchBuilder::AVX128_VPMULHW<true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE7), 1, &OpDispatchBuilder::AVX128_MOVVectorNT},
|
||||
|
||||
@@ -1058,18 +1058,8 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs) {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSizeFromSrc(Op), Op->Flags);
|
||||
}
|
||||
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Result {};
|
||||
if constexpr (HostRoundingMode) {
|
||||
Result = _Float_ToGPR_S(GPRSize, SrcElementSize, Src.Low);
|
||||
} else {
|
||||
Result = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src.Low);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src.Low, SrcElementSize, HostRoundingMode);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
|
||||
@@ -1604,7 +1594,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
@@ -1614,46 +1604,20 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128BitSrc);
|
||||
RefPair Result {};
|
||||
|
||||
if (SrcElementSize == OpSize::i64Bit && Narrow) {
|
||||
///< Special case for VCVTPD2DQ/CVTTPD2DQ because it has weird rounding requirements.
|
||||
Result.Low = _Vector_F64ToI32(OpSize::i128Bit, Src.Low, HostRoundingMode ? Round_Host : Round_Towards_Zero, Is128BitSrc);
|
||||
|
||||
if (!Is128BitSrc) {
|
||||
// Also convert the upper 128-bit lane
|
||||
auto ResultHigh = _Vector_F64ToI32(OpSize::i128Bit, Src.High, HostRoundingMode ? Round_Host : Round_Towards_Zero, false);
|
||||
|
||||
// Zip the two halves together in to the lower 128-bits
|
||||
Result.Low = _VZip(OpSize::i128Bit, OpSize::i64Bit, Result.Low, ResultHigh);
|
||||
}
|
||||
} else {
|
||||
auto Convert = [this](Ref Src) -> Ref {
|
||||
auto ElementSize = SrcElementSize;
|
||||
if (Narrow) {
|
||||
ElementSize = ElementSize >> 1;
|
||||
Src = _Vector_FToF(OpSize::i128Bit, ElementSize, Src, SrcElementSize);
|
||||
}
|
||||
|
||||
if (HostRoundingMode) {
|
||||
return _Vector_FToS(OpSize::i128Bit, ElementSize, Src);
|
||||
} else {
|
||||
return _Vector_FToZS(OpSize::i128Bit, ElementSize, Src);
|
||||
}
|
||||
};
|
||||
|
||||
Result.Low = Convert(Src.Low);
|
||||
|
||||
if (!Is128BitSrc) {
|
||||
if (!Narrow) {
|
||||
Result.High = Convert(Src.High);
|
||||
} else {
|
||||
Result.Low = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 0, Result.Low, Convert(Src.High));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Narrow || Is128BitSrc) {
|
||||
Result.Low = Vector_CVT_Float_To_Int32Impl(Op, OpSize::i128Bit, Src.Low, OpSize::i128Bit, SrcElementSize, HostRoundingMode, Is128BitSrc);
|
||||
if (Is128BitSrc) {
|
||||
// Zero the upper 128-bit lane of the result.
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
} else {
|
||||
Result.High = Vector_CVT_Float_To_Int32Impl(Op, OpSize::i128Bit, Src.High, OpSize::i128Bit, SrcElementSize, HostRoundingMode, false);
|
||||
// Also convert the upper 128-bit lane
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
// Zip the two halves together in to the lower 128-bits
|
||||
Result.Low = _VZip(OpSize::i128Bit, OpSize::i64Bit, Result.Low, Result.High);
|
||||
|
||||
// Zero the upper 128-bit lane of the result.
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
}
|
||||
}
|
||||
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
|
||||
@@ -7,7 +7,7 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
|
||||
{0x0D, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
|
||||
@@ -19,7 +19,7 @@ $end_info$
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
FEXCore::X86State::RFLAG_CF_RAW_LOC, FEXCore::X86State::RFLAG_PF_RAW_LOC, FEXCore::X86State::RFLAG_AF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_RAW_LOC, FEXCore::X86State::RFLAG_SF_RAW_LOC, FEXCore::X86State::RFLAG_TF_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_RAW_LOC, FEXCore::X86State::RFLAG_SF_RAW_LOC, FEXCore::X86State::RFLAG_TF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_IF_LOC, FEXCore::X86State::RFLAG_DF_RAW_LOC, FEXCore::X86State::RFLAG_OF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_IOPL_LOC, FEXCore::X86State::RFLAG_NT_LOC, FEXCore::X86State::RFLAG_RF_LOC,
|
||||
FEXCore::X86State::RFLAG_VM_LOC, FEXCore::X86State::RFLAG_AC_LOC, FEXCore::X86State::RFLAG_VIF_LOC,
|
||||
@@ -185,8 +185,9 @@ Ref OpDispatchBuilder::LoadAF() {
|
||||
// Read the result, stored for PF.
|
||||
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
// What's left is to XOR and extract. This is the deferred part.
|
||||
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i32Bit, AFWord, Result));
|
||||
// What's left is to XOR and extract. This is the deferred part. We
|
||||
// specifically use a 64-bit Xor here as we don't need masking.
|
||||
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i64Bit, AFWord, Result));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FixupAF() {
|
||||
@@ -199,7 +200,8 @@ void OpDispatchBuilder::FixupAF() {
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
Ref XorRes = _Xor(OpSize::i32Bit, AFRaw, PFRaw);
|
||||
// Again 64-bit as masking is more expensive given our ConstProp design.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, AFRaw, PFRaw);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
@@ -238,8 +240,8 @@ void OpDispatchBuilder::CalculateAF(Ref Src1, Ref Src2) {
|
||||
|
||||
// We store the XOR of the arguments. At read time, we XOR with the
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit.
|
||||
Ref XorRes = _Xor(OpSize::i32Bit, Src1, Src2);
|
||||
// appropriate bit. Again 64-bit to avoid masking.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, Src1, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
|
||||
@@ -6,40 +6,66 @@ namespace FEXCore::IR {
|
||||
#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode)
|
||||
#define PF_3A_NONE 0
|
||||
#define PF_3A_66 1
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<OpSize::i16Bit>},
|
||||
constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
constexpr auto OpDispatchTableGenH0F3AREX = []<uint16_t REX>() consteval {
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> Table[] = {
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<OpSize::i16Bit>},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(0, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(REX, PF_3A_NONE, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(REX, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i8Bit>},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i8Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<OpSize::i8Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x21), 1, &OpDispatchBuilder::InsertPSOp},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
return std::to_array(Table);
|
||||
};
|
||||
|
||||
auto REX0 = OpDispatchTableGenH0F3AREX.template operator()<0>();
|
||||
auto REX1 = OpDispatchTableGenH0F3AREX.template operator()<1>();
|
||||
auto concat = []<typename T, size_t N1, size_t N2>(std::array<T, N1> const& lhs,
|
||||
std::array<T, N2> const& rhs) consteval -> std::array<T, N1 + N2> {
|
||||
std::array<T, N1 + N2> Table {};
|
||||
for (size_t i = 0; i < N1; ++i) {
|
||||
Table[i] = lhs[i];
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < N2; ++i) {
|
||||
Table[N1 + i] = rhs[i];
|
||||
}
|
||||
|
||||
return Table;
|
||||
};
|
||||
return concat(REX0, REX1);
|
||||
};
|
||||
|
||||
constexpr auto OpDispatch_H0F3ATableIgnoreREX = OpDispatchTableGenH0F3A();
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATableNeedsREX0[] = {
|
||||
{OPD(0, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<OpSize::i8Bit>},
|
||||
{OPD(0, PF_3A_66, 0x21), 1, &OpDispatchBuilder::InsertPSOp},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(1, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i64Bit>},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i64Bit>},
|
||||
};
|
||||
|
||||
@@ -57,8 +57,8 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i32Bit>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i32Bit>},
|
||||
@@ -161,7 +161,7 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
|
||||
@@ -200,7 +200,7 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i64Bit>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
{0xF0, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
};
|
||||
|
||||
@@ -213,8 +213,8 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
|
||||
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i64Bit>},
|
||||
@@ -226,7 +226,7 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i64Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i64Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i64Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i64Bit>},
|
||||
@@ -284,7 +284,7 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
|
||||
@@ -2067,6 +2067,24 @@ void OpDispatchBuilder::AVXCVTGPR_To_FPR(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::AVXCVTGPR_To_FPR<OpSize::i32Bit>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXCVTGPR_To_FPR<OpSize::i64Bit>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElementSize, bool HostRoundingMode) {
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcElementSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
Ref Converted = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
|
||||
bool Dst32 = GPRSize == OpSize::i32Bit;
|
||||
Ref MaxI = Dst32 ? _Constant(0x80000000) : _Constant(0x8000000000000000);
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcElementSize, (SrcElementSize == OpSize::i32Bit) ?
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F32_I32 : NAMED_VECTOR_CVTMAX_F32_I64) :
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F64_I32 : NAMED_VECTOR_CVTMAX_F64_I64));
|
||||
return _Select(GPRSize, SrcElementSize, CondClassType {FEXCore::IR::COND_FGT}, MaxF, Src, Converted, MaxI);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) {
|
||||
// If loading a vector, use the full size, so we don't
|
||||
@@ -2074,18 +2092,8 @@ void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) {
|
||||
// memory, then we want to load the element size exactly.
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op);
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
if constexpr (HostRoundingMode) {
|
||||
Src = _Float_ToGPR_S(GPRSize, SrcElementSize, Src);
|
||||
} else {
|
||||
Src = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, GPRSize, OpSize::iInvalid);
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src, SrcElementSize, HostRoundingMode);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
@@ -2127,77 +2135,43 @@ void OpDispatchBuilder::Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Widen>
|
||||
void OpDispatchBuilder::AVXVector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
Ref Result = Vector_CVT_Int_To_FloatImpl(Op, SrcElementSize, Widen);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Int_To_Float<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Int_To_Float<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::Vector_CVT_Float_To_IntImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
auto ElementSize = SrcElementSize;
|
||||
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
if (Narrow) {
|
||||
Src = _Vector_FToF(DstSize, SrcElementSize >> 1, Src, SrcElementSize);
|
||||
ElementSize = ElementSize >> 1;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::Vector_CVT_Float_To_Int32Impl(OpcodeArgs, IR::OpSize DstSize, Ref Src, IR::OpSize SrcSize, IR::OpSize SrcElementSize,
|
||||
bool HostRoundingMode, bool ZeroUpperHalf) {
|
||||
if (HostRoundingMode) {
|
||||
return _Vector_FToS(DstSize, ElementSize, Src);
|
||||
} else {
|
||||
return _Vector_FToZS(DstSize, ElementSize, Src);
|
||||
Src = _Vector_FToI(SrcSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
OpSize OverflowConstSize = ZeroUpperHalf && SrcElementSize == OpSize::i64Bit ? DstSize / 2 : DstSize;
|
||||
Ref MaxI = LoadAndCacheNamedVectorConstant(OverflowConstSize, NAMED_VECTOR_CVTMAX_I32);
|
||||
Ref Converted {}, Cmp {};
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_CVTMAX_F64_I32);
|
||||
Converted = _Vector_F64ToI32(DstSize, Src, Round_Towards_Zero, ZeroUpperHalf);
|
||||
|
||||
Cmp = _VFCMPGT(SrcSize, OpSize::i64Bit, MaxF, Src);
|
||||
Cmp = _VUShrNI(DstSize, OpSize::i64Bit, Cmp, 32);
|
||||
} else {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(DstSize, NAMED_VECTOR_CVTMAX_F32_I32);
|
||||
Converted = _Vector_FToZS(DstSize, OpSize::i32Bit, Src);
|
||||
Cmp = _VFCMPGT(DstSize, OpSize::i32Bit, MaxF, Src);
|
||||
}
|
||||
return _VBSL(DstSize, Cmp, Converted, MaxI);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Result {};
|
||||
if (SrcElementSize == OpSize::i64Bit && Narrow) {
|
||||
///< Special case for CVTTPD2DQ because it has weird rounding requirements.
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Result = _Vector_F64ToI32(DstSize, Src, HostRoundingMode ? Round_Host : Round_Towards_Zero, true);
|
||||
} else {
|
||||
Result = Vector_CVT_Float_To_IntImpl(Op, SrcElementSize, Narrow, HostRoundingMode);
|
||||
}
|
||||
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = Vector_CVT_Float_To_Int32Impl(Op, DstSize, Src, OpSizeFromSrc(Op), SrcElementSize, HostRoundingMode, true);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::AVXVector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Result {};
|
||||
if (SrcElementSize == OpSize::i64Bit && Narrow) {
|
||||
///< Special case for CVTPD2DQ/CVTTPD2DQ because it has weird rounding requirements.
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Result = _Vector_F64ToI32(DstSize, Src, HostRoundingMode ? Round_Host : Round_Towards_Zero, true);
|
||||
} else {
|
||||
Result = Vector_CVT_Float_To_IntImpl(Op, SrcElementSize, Narrow, HostRoundingMode);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i32Bit, false, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i32Bit, false, true>(OpcodeArgs);
|
||||
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i64Bit, true, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVector_CVT_Float_To_Int<OpSize::i64Bit, true, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op) {
|
||||
@@ -2277,7 +2251,7 @@ void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Src, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
// This function causes a change in MMX state from X87 to MMX
|
||||
if (MMXState == MMXState_X87) {
|
||||
@@ -2288,29 +2262,16 @@ void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
// unnecessarily zero extend the vector. Otherwise, if
|
||||
// memory, then we want to load the element size exactly.
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op);
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
auto ElementSize = SrcElementSize;
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
|
||||
if (Narrow) {
|
||||
Src = _Vector_FToF(Size, SrcElementSize >> 1, Src, SrcElementSize);
|
||||
ElementSize = ElementSize >> 1;
|
||||
}
|
||||
|
||||
if constexpr (HostRoundingMode) {
|
||||
Src = _Vector_FToS(Size, ElementSize, Src);
|
||||
} else {
|
||||
Src = _Vector_FToZS(Size, ElementSize, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, OpSize::iInvalid);
|
||||
Ref Result = Vector_CVT_Float_To_Int32Impl(Op, DstSize, Src, SrcSize, SrcElementSize, HostRoundingMode, false /* TODO? */);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
@@ -609,8 +609,8 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
uint8_t Offset = Op->OP & 7;
|
||||
Res = _F80CmpStack(Offset);
|
||||
} else {
|
||||
// Memory arg
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, Width);
|
||||
@@ -618,6 +618,8 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, Width);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
Res = _F80CmpValue(b);
|
||||
}
|
||||
|
||||
@@ -157,6 +157,8 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -193,6 +195,8 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -244,6 +248,8 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -299,6 +305,8 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -330,22 +338,22 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
|
||||
// Implicit arg
|
||||
uint8_t offset = Op->OP & 7;
|
||||
b = _ReadStackValue(offset);
|
||||
} else {
|
||||
} else if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
if (WhichFlags == FCOMIFlags::FLAGS_X87) {
|
||||
|
||||
@@ -145,7 +145,7 @@ std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
// These three are all X87 instructions
|
||||
{0x9B, 1, X86InstInfo{"FWAIT", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9C, 1, X86InstInfo{"PUSHF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF), 0, nullptr}},
|
||||
{0x9D, 1, X86InstInfo{"POPF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF), 0, nullptr}},
|
||||
{0x9D, 1, X86InstInfo{"POPF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -21,49 +21,60 @@ constexpr uint16_t PF_3A_66 = 1;
|
||||
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> Table{};
|
||||
constexpr U16U8InfoStruct H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
auto TableGen = []<uint16_t REX>() consteval {
|
||||
constexpr U16U8InfoStruct Table[] = {
|
||||
{OPD(REX, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
};
|
||||
return std::to_array(Table);
|
||||
};
|
||||
constexpr auto H0F3ATable_IgnoresREX0 = TableGen.template operator()<0>();
|
||||
constexpr auto H0F3ATable_IgnoresREX1 = TableGen.template operator()<1>();
|
||||
|
||||
GenerateTable(&Table.at(0), H0F3ATable, std::size(H0F3ATable));
|
||||
GenerateTable(&Table.at(0), &H0F3ATable_IgnoresREX0.at(0), H0F3ATable_IgnoresREX0.size());
|
||||
GenerateTable(&Table.at(0), &H0F3ATable_IgnoresREX1.at(0), H0F3ATable_IgnoresREX1.size());
|
||||
|
||||
constexpr U16U8InfoStruct TableNeedsREX[] = {
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
};
|
||||
GenerateTable(&Table.at(0), TableNeedsREX, std::size(TableNeedsREX));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATableIgnoreREX);
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATableNeedsREX0);
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATable);
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
static constexpr U16U8InfoStruct H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRQ", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
};
|
||||
|
||||
+161
-142
@@ -7,6 +7,7 @@
|
||||
" SSA = untyped",
|
||||
" GPR = GPR class type",
|
||||
" FPR = FPR class type",
|
||||
" PRED = Predicate register class type",
|
||||
"Declaring the SSA types correctly will allow validation passes to ensure the op is getting passed correct arguments",
|
||||
"",
|
||||
"Arguments must always follow a particular order. <Type>:<Prefix><Name>",
|
||||
@@ -83,6 +84,7 @@
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {2}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRFixedClass {3}",
|
||||
"constexpr FEXCore::IR::RegisterClassType PREDClass {4}",
|
||||
"constexpr FEXCore::IR::RegisterClassType ComplexClass {5}",
|
||||
"constexpr FEXCore::IR::RegisterClassType InvalidClass {7}",
|
||||
"",
|
||||
@@ -148,6 +150,7 @@
|
||||
"SSA": "OrderedNode*",
|
||||
"GPR": "OrderedNode*",
|
||||
"FPR": "OrderedNode*",
|
||||
"PRED": "OrderedNode*",
|
||||
"FenceType": "FenceType",
|
||||
"RegisterClass": "RegisterClassType",
|
||||
"CondClass": "CondClassType",
|
||||
@@ -258,7 +261,7 @@
|
||||
"FPR = AllocateFPR OpSize:#RegisterSize, OpSize:#ElementSize": {
|
||||
"Desc": ["Like AllocateGPR, but for FPR"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"GPR = AllocateGPRAfter GPR:$After": {
|
||||
"Desc": ["Silly pseudo-instruction to allocate a register for a future destination",
|
||||
@@ -560,11 +563,27 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value1) == $Class",
|
||||
"WalkFindRegClass($Value2) == $Class"
|
||||
"WalkFindRegClass($Value1) == $Class"
|
||||
]
|
||||
},
|
||||
|
||||
"PRED = InitPredicate OpSize:#Size, u8:$Pattern": {
|
||||
"Desc": ["Initialize predicate register from Pattern"],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StoreMemPredicate OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Value, PRED:$Mask, GPR:$Addr": {
|
||||
"Desc": [ "Stores a value to memory using SVE predicate mask." ],
|
||||
"DestSize": "RegisterSize",
|
||||
"HasSideEffects": true,
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = LoadMemPredicate OpSize:#RegisterSize, OpSize:#ElementSize, PRED:$Mask, GPR:$Addr": {
|
||||
"Desc": [ "Loads a value to memory using SVE predicate mask." ],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"SSA = LoadMemTSO RegisterClass:$Class, OpSize:#Size, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a x86 TSO compatible load from memory. Offset must be Invalid()."
|
||||
],
|
||||
@@ -588,7 +607,7 @@
|
||||
"determines whether or not that element will be loaded from memory"],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"VStoreVectorMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Mask, FPR:$Data, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a masked store similar to VPMASKMOV/VMASKMOV where the upper bit of each element",
|
||||
@@ -596,7 +615,7 @@
|
||||
"HasSideEffects": true,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VLoadVectorGatherMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$Mask, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, OpSize:$VectorIndexElementSize, u8:$OffsetScale, u8:$DataElementOffsetStart, u8:$IndexElementOffsetStart": {
|
||||
"Desc": [
|
||||
@@ -607,7 +626,7 @@
|
||||
"TiedSource": 0,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"$VectorIndexElementSize == OpSize::i32Bit || $VectorIndexElementSize == OpSize::i64Bit"
|
||||
]
|
||||
@@ -622,7 +641,7 @@
|
||||
"TiedSource": 0,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize == OpSize::i32Bit",
|
||||
"RegisterSize != FEXCore::IR::OpSize::i256Bit && \"What does 256-bit mean in this context?\""
|
||||
@@ -634,19 +653,19 @@
|
||||
"Matches arm64 ld1 semantics"],
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"VStoreVectorElement OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Value, u8:$Index, GPR:$Addr": {
|
||||
"Desc": ["Does a memory store of a single element of a vector.",
|
||||
"Matches arm64 st1 semantics"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VBroadcastFromMem OpSize:#RegisterSize, OpSize:#ElementSize, GPR:$Address": {
|
||||
"Desc": ["Broadcasts an ElementSize value from memory into each element of a vector."],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"GPR = Push OpSize:#Size, OpSize:$ValueSize, GPR:$Value, GPR:$Addr": {
|
||||
"Desc": [
|
||||
@@ -1685,7 +1704,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFSubScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'sub' between Vector1 and Vector2.",
|
||||
@@ -1695,7 +1714,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFMulScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'mul' between Vector1 and Vector2.",
|
||||
@@ -1705,7 +1724,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFDivScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'div' between Vector1 and Vector2.",
|
||||
@@ -1715,7 +1734,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFMinScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'min' between Vector1 and Vector2.",
|
||||
@@ -1728,7 +1747,7 @@
|
||||
"If either source operand is NaN then return the second operand."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"ImplicitFlagClobber": true
|
||||
},
|
||||
"FPR = VFMaxScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
@@ -1742,7 +1761,7 @@
|
||||
"If either source operand is NaN then return the second operand."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"ImplicitFlagClobber": true
|
||||
},
|
||||
"FPR = VFSqrtScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
@@ -1753,7 +1772,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFRSqrtScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'rsqrt' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
@@ -1763,7 +1782,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFRecpScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'recip' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
@@ -1773,7 +1792,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFToFScalarInsert OpSize:#RegisterSize, OpSize:#DstElementSize, OpSize:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cvt' between Vector1 and Vector2.",
|
||||
@@ -1783,7 +1802,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
"ElementSize": "DstElementSize"
|
||||
},
|
||||
"FPR = VSToFVectorInsert OpSize:#RegisterSize, OpSize:#DstElementSize, OpSize:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i8:$HasTwoElements, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a Vector 'scvt' between Vector1 and Vector2.",
|
||||
@@ -1795,7 +1814,7 @@
|
||||
"Handles the edge case of cvtpi2ps xmm0, mm0 which is two elements in the lower 64-bits"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
"ElementSize": "DstElementSize"
|
||||
},
|
||||
"FPR = VSToFGPRInsert OpSize:#RegisterSize, OpSize:#DstElementSize, OpSize:$SrcElementSize, FPR:$Vector, GPR:$Src, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cvt' between Vector1 and GPR.",
|
||||
@@ -1805,7 +1824,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
"ElementSize": "DstElementSize"
|
||||
},
|
||||
"FPR = VFToIScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, RoundType:$Round, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar round float to integral on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
@@ -1816,7 +1835,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, FloatCompareOp:$Op, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cmp' between Vector1 and Vecto2, inserting in to Vector1 and storing in to the destination.",
|
||||
@@ -1827,7 +1846,7 @@
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFMLAScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
@@ -1836,7 +1855,7 @@
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFMLSScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
@@ -1846,7 +1865,7 @@
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFNMLAScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
@@ -1856,7 +1875,7 @@
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFNMLSScalarInsert OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
@@ -1866,7 +1885,7 @@
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
}
|
||||
},
|
||||
@@ -1883,7 +1902,7 @@
|
||||
"Desc": ["Generates a vector with each element containg the immediate zexted"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = LoadNamedVectorConstant OpSize:#RegisterSize, NamedVectorConstant:$Constant": {
|
||||
@@ -1901,25 +1920,25 @@
|
||||
},
|
||||
"FPR = VNeg OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VNot OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VAbs OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Does an signed integer absolute"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VPopcount OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Does a popcount for each element of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VAddV OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
@@ -1927,49 +1946,49 @@
|
||||
"Result is a zero extended scalar"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUMinV OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Does a horizontal vector unsigned minimum of elements across the source vector",
|
||||
"Result is a zero extended scalar"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUMaxV OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Does a horizontal vector unsigned maximum of elements across the source vector",
|
||||
"Result is a zero extended scalar"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFAbs OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VFNeg OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VFRecp OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VFSqrt OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VFRSqrt OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VCMPEQZ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VCMPGTZ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Vector compare signed greater than",
|
||||
@@ -1977,7 +1996,7 @@
|
||||
"Compares the vector against zero"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VCMPLTZ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Vector compare signed less than",
|
||||
@@ -1985,39 +2004,39 @@
|
||||
"Compares the vector against zero"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VDupElement OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$Index": {
|
||||
"Desc": ["Duplicates one element from the source register across the whole register"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VShlI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUShrI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUShraI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VSShrI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VUShrNI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"Desc": "Unsigned shifts right each element and then narrows to the next lower element size",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
},
|
||||
|
||||
"FPR = VUShrNI2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper, u8:$BitShift": {
|
||||
@@ -2026,73 +2045,73 @@
|
||||
"Inserts results in to the high elements of the first argument"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
},
|
||||
"FPR = VSXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Sign extends elements from the source element size to the next size up",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSXTL2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Sign extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSSHLL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift{0}": {
|
||||
"Desc": "Sign extends elements from the source element size to the next size up",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSSHLL2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift{0}": {
|
||||
"Desc": ["Sign extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Zero extends elements from the source element size to the next size up",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUXTL2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Zero extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSQXTN OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
},
|
||||
"FPR = VSQXTN2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
},
|
||||
"FPR = VSQXTNPair OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"Desc": ["Does both VSQXTN and VSQXTN2 in a combined operation."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
},
|
||||
"FPR = VSQXTUN OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
},
|
||||
"FPR = VSQXTUN2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
},
|
||||
"FPR = VSQXTUNPair OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"Desc": ["Does both VSQXTUN and VSQXTUN2 in a combined operation."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
},
|
||||
"FPR = VSRSHR OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"Desc": ["Signed rounding shift right by immediate",
|
||||
@@ -2100,7 +2119,7 @@
|
||||
],
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VSQSHL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"Desc": ["Signed satuating shift left by immediate",
|
||||
@@ -2108,265 +2127,265 @@
|
||||
],
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VRev32 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc" : ["Reverses elements in 32-bit halfwords",
|
||||
"Available element size: 1byte, 2 byte"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VRev64 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc" : ["Reverses elements in 64-bit halfwords",
|
||||
"Available element size: 1byte, 2 byte, 4 byte"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VAdd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VSub OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VAnd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VAndn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VOr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VXor OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VUQAdd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VUQSub OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VSQAdd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VSQSub OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VAddP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"Desc": "Does a horizontal pairwise add of elements across the two source vectors",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VURAvg OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Does an unsigned rounded average", "dst_elem = (src1_elem + src2_elem + 1) >> 1"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUMin OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUMax OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VSMin OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VSMax OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VZip OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VZip2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUnZip OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUnZip2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VTrn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VTrn2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VFAdd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFAddP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"Desc": "Does a horizontal pairwise add of elements across the two source vectors with float element types",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFAddV OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Does a horizontal float vector add of elements across the source vector",
|
||||
"Result is a zero extended scalar"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFSub OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFMul OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFDiv OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VFMin OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFMax OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VMul OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUMull OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSMull OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": [ "Does a signed integer multiply with extend.",
|
||||
"ElementSize is the source size"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUMull2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": "Multiplies the high elements with size extension",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSMull2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": "Multiplies the high elements with size extension",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": "Wide unsigned multiply returning the high results",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VSMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": "Wide signed multiply returning the high results",
|
||||
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUABDL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Unsigned Absolute Difference Long"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUABDL2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Unsigned Absolute Difference Long",
|
||||
"Using the high elements of the source vectors"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUShl OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, FPR:$ShiftVector, i1:$RangeCheck": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUShr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, FPR:$ShiftVector, i1:$RangeCheck": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VSShr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, FPR:$ShiftVector, i1:$RangeCheck": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUShlS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUShrS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VSShrS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUShrSWide OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VSShrSWide OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUShlSWide OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VInsElement OpSize:#RegisterSize, OpSize:#ElementSize, u8:$DestIdx, u8:$SrcIdx, FPR:$DestVector, FPR:$SrcVector": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VInsGPR OpSize:#RegisterSize, OpSize:#ElementSize, u8:$DestIdx, FPR:$DestVector, GPR:$Src": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VExtr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper, u8:$Index": {
|
||||
@@ -2377,12 +2396,12 @@
|
||||
"Dest = TmpVector >> (ElementSize * Index * 8); // Or can be thought of `concat(&TmpVector[Index], i128)`"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VCMPEQ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VCMPGT OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
@@ -2391,35 +2410,35 @@
|
||||
],
|
||||
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPEQ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPNEQ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPLT OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPGT OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPLE OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPORD OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPUNO OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VTBL1 OpSize:#RegisterSize, FPR:$VectorTable, FPR:$VectorIndices": {
|
||||
"Desc": ["Does a vector table lookup from one register in to the destination",
|
||||
@@ -2484,7 +2503,7 @@
|
||||
},
|
||||
"FPR = VFCADD OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, u16:$Rotate": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFMLA OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
@@ -2492,7 +2511,7 @@
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 2
|
||||
},
|
||||
"FPR = VFMLS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
@@ -2501,7 +2520,7 @@
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 2
|
||||
},
|
||||
"FPR = VFNMLA OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
@@ -2510,7 +2529,7 @@
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 2
|
||||
},
|
||||
"FPR = VFNMLS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
@@ -2519,7 +2538,7 @@
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 2
|
||||
}
|
||||
},
|
||||
@@ -2529,13 +2548,13 @@
|
||||
"No conversion is done on the data as it moves register files"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VDupFromGPR OpSize:#RegisterSize, OpSize:#ElementSize, GPR:$Src": {
|
||||
"Desc": ["Broadcasts a value in a GPR into each ElementSize-sized element in a vector"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = Float_FromGPR_S OpSize:#DstElementSize, OpSize:$SrcElementSize, GPR:$Src": {
|
||||
@@ -2554,24 +2573,24 @@
|
||||
"FPR = Vector_SToF OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Vector op: Converts signed integer to same size float",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_FToS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Vector op: Converts float to signed integer, rounding towards zero",
|
||||
"Rounding mode determined by host rounding mode"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_FToZS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Vector op: Converts float to signed integer, rounding towards zero",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_FToF OpSize:#RegisterSize, OpSize:#DestElementSize, FPR:$Vector, OpSize:$SrcElementSize": {
|
||||
"Desc": "Vector op: Converts float from source element size to destination size (fp32<->fp64)",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DestElementSize"
|
||||
"ElementSize": "DestElementSize"
|
||||
},
|
||||
|
||||
"FPR = VFCVTL2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
@@ -2580,7 +2599,7 @@
|
||||
"Selecting from the high half of the register."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)",
|
||||
"ElementSize": "ElementSize << 1",
|
||||
"EmitValidation": [
|
||||
"RegisterSize != FEXCore::IR::OpSize::i256Bit && \"What does 256-bit mean in this context?\""
|
||||
]
|
||||
@@ -2594,7 +2613,7 @@
|
||||
"F64->F32, F32->F16"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)",
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
"RegisterSize != FEXCore::IR::OpSize::i256Bit && \"What does 256-bit mean in this context?\""
|
||||
]
|
||||
@@ -2604,14 +2623,14 @@
|
||||
"Rounding mode determined by argument"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "IR::NumElements(RegisterSize, ElementSize)"
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_F64ToI32 OpSize:#RegisterSize, FPR:$Vector, RoundType:$Round, i1:$EnsureZeroUpperHalf": {
|
||||
"Desc": ["Vector op: Rounds 64-bit float to 32-bit integral with round mode",
|
||||
"Matches CVTPD2DQ/CVTTPD2DQ behaviour"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / FEXCore::IR::OpSize::i32Bit"
|
||||
"ElementSize": "FEXCore::IR::OpSize::i32Bit"
|
||||
}
|
||||
},
|
||||
"Crypto": {
|
||||
|
||||
@@ -77,6 +77,8 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
*out << "FPR";
|
||||
} else if (Arg == FPRFixedClass.Val) {
|
||||
*out << "FPRFixed";
|
||||
} else if (Arg == PREDClass.Val) {
|
||||
*out << "PRED";
|
||||
} else {
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
@@ -98,6 +100,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::PREDClass.Val: *out << "(PRED"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
@@ -206,6 +209,22 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
return "x87_log10_2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_LOG_2:
|
||||
return "x87_log2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I32:
|
||||
return "cvtmax_f32_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I32_UPPER:
|
||||
return "cvtmax_f32_i32_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I64:
|
||||
return "cvtmax_f32_i64";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I32:
|
||||
return "cvtmax_f64_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I32_UPPER:
|
||||
return "cvtmax_f64_i32_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I64:
|
||||
return "cvtmax_f64_i64";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I32:
|
||||
return "cvtmax_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I64:
|
||||
return "cvtmax_i64";
|
||||
default:
|
||||
return "<Unknown Named Vector Constant>";
|
||||
}
|
||||
|
||||
@@ -41,6 +41,7 @@ FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
case FPRClass:
|
||||
case GPRFixedClass:
|
||||
case FPRFixedClass:
|
||||
case PREDClass:
|
||||
case InvalidClass: return Class;
|
||||
default: break;
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
|
||||
@@ -9,9 +10,9 @@
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <new>
|
||||
#include <stdint.h>
|
||||
#include <string.h>
|
||||
|
||||
@@ -45,6 +46,37 @@ public:
|
||||
}
|
||||
void ResetWorkingList();
|
||||
|
||||
// Predicate Cache Implementation
|
||||
// This lives here rather than OpcodeDispatcher because x87StackOptimization Pass
|
||||
// also needs it.
|
||||
struct PredicateKey {
|
||||
ARMEmitter::PredicatePattern Pattern;
|
||||
OpSize Size;
|
||||
bool operator==(const PredicateKey& rhs) const = default;
|
||||
};
|
||||
|
||||
struct PredicateKeyHash {
|
||||
size_t operator()(const PredicateKey& key) const {
|
||||
return FEXCore::ToUnderlying(key.Pattern) + (FEXCore::ToUnderlying(key.Size) * FEXCore::ToUnderlying(OpSize::iInvalid));
|
||||
}
|
||||
};
|
||||
fextl::unordered_map<PredicateKey, Ref, PredicateKeyHash> InitPredicateCache;
|
||||
|
||||
Ref InitPredicateCached(OpSize Size, ARMEmitter::PredicatePattern Pattern) {
|
||||
PredicateKey Key {Pattern, Size};
|
||||
auto ValIt = InitPredicateCache.find(Key);
|
||||
if (ValIt == InitPredicateCache.end()) {
|
||||
auto Predicate = _InitPredicate(Size, static_cast<uint8_t>(FEXCore::ToUnderlying(Pattern)));
|
||||
InitPredicateCache[Key] = Predicate;
|
||||
return Predicate;
|
||||
}
|
||||
return ValIt->second;
|
||||
}
|
||||
|
||||
void ResetInitPredicateCache() {
|
||||
InitPredicateCache.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* @name IR allocation routines
|
||||
*
|
||||
|
||||
@@ -70,7 +70,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass());
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
|
||||
@@ -5,7 +5,8 @@
|
||||
|
||||
namespace FEXCore {
|
||||
class CPUIDEmu;
|
||||
}
|
||||
struct HostFeatures;
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
class IntrusivePooledAllocator;
|
||||
@@ -19,7 +20,7 @@ class RegisterAllocationData;
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&);
|
||||
|
||||
namespace Validation {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
|
||||
@@ -18,13 +18,8 @@ $end_info$
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <string.h>
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
@@ -188,6 +183,35 @@ void ConstProp::HandleConstantPools(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
}
|
||||
|
||||
// Helper to replace the destination of an instruction with one of its sources,
|
||||
// to implement algebraic identities. This is surprisingly tricky due to
|
||||
// implicit masking in our IR.
|
||||
//
|
||||
// FEX's IR uses sized opcodes, matching arm64 semantics. 64-bit opcodes do not
|
||||
// mask, whereas smaller opcodes mask/zero-extend from 32-bits. Therefore, if
|
||||
// the instruction is 32-bit, we need to mask the source for a sound
|
||||
// replacement, in case there was garbage in the upper bits.
|
||||
//
|
||||
// However, if that source is in turn written by a 32-bit instruction, it is
|
||||
// guaranteed to have already been masked, so we know there's no garbage and we
|
||||
// can avoid the zero-extension. This is the case 99% of the time, but the
|
||||
// masking here is correctness-bearing nevertheless (and new versions of Denuvo
|
||||
// break if you get this wrong!)
|
||||
static inline void ReplaceWithSource(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Idx) {
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[Idx]);
|
||||
|
||||
if (IROp->Size < OpSize::i64Bit) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == OpSize::i32Bit, "other sizes not here");
|
||||
|
||||
auto Header = IREmit->GetOpHeader(IROp->Args[Idx]);
|
||||
if (Header->Size > OpSize::i32Bit) {
|
||||
Arg = IREmit->_Bfe(OpSize::i32Bit, 32, 0, Arg);
|
||||
}
|
||||
}
|
||||
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
}
|
||||
|
||||
// constprop + some more per instruction logic
|
||||
void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp) {
|
||||
switch (IROp->Op) {
|
||||
@@ -285,7 +309,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
Replaced = true;
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID() || (Constant2 & getMask(IROp)) == getMask(IROp)) {
|
||||
// AND with same value results in original value
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
Replaced = true;
|
||||
}
|
||||
|
||||
@@ -318,8 +342,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[1 - i]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 1 - i);
|
||||
Replaced = true;
|
||||
break;
|
||||
}
|
||||
@@ -361,8 +384,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[0]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
@@ -373,8 +395,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[0]);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
|
||||
@@ -47,6 +47,7 @@ struct RegState {
|
||||
// On arm64, there are 16 Fixed and 12 normal
|
||||
FPRsFixed[Reg.Reg] = ssa;
|
||||
return true;
|
||||
case PREDClass: PREGs[Reg.Reg] = ssa; return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -59,6 +60,7 @@ struct RegState {
|
||||
case GPRFixedClass: return GPRsFixed[Reg.Reg];
|
||||
case FPRClass: return FPRs[Reg.Reg];
|
||||
case FPRFixedClass: return FPRsFixed[Reg.Reg];
|
||||
case PREDClass: return PREGs[Reg.Reg];
|
||||
}
|
||||
return InvalidReg;
|
||||
}
|
||||
@@ -82,6 +84,7 @@ private:
|
||||
std::array<IR::NodeID, 32> FPRsFixed = {};
|
||||
std::array<IR::NodeID, 32> GPRs = {};
|
||||
std::array<IR::NodeID, 32> FPRs = {};
|
||||
std::array<IR::NodeID, 32> PREGs = {};
|
||||
|
||||
fextl::unordered_map<uint32_t, IR::NodeID> Spills;
|
||||
};
|
||||
|
||||
@@ -3,9 +3,10 @@
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/deque.h>
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/Profiler.h"
|
||||
#include "FEXCore/Core/HostFeatures.h"
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
@@ -146,13 +147,15 @@ private:
|
||||
|
||||
class X87StackOptimization final : public Pass {
|
||||
public:
|
||||
X87StackOptimization() {
|
||||
X87StackOptimization(const FEXCore::HostFeatures& Features)
|
||||
: Features(Features) {
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
ReducedPrecisionMode = ReducedPrecision;
|
||||
}
|
||||
void Run(IREmitter* Emit) override;
|
||||
|
||||
private:
|
||||
const FEXCore::HostFeatures& Features;
|
||||
bool ReducedPrecisionMode;
|
||||
|
||||
// Helpers
|
||||
@@ -820,11 +823,16 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
}
|
||||
if (Op->StoreSize == OpSize::f80Bit) { // Part of code from StoreResult_WithOpSize()
|
||||
// For X87 extended doubles, split before storing
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
auto DestAddr = IREmit->_Add(OpSize::i64Bit, AddrNode, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, DestAddr, Upper, OpSize::i64Bit);
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
auto PReg = IREmit->InitPredicateCached(OpSize::i16Bit, ARMEmitter::PredicatePattern::SVE_VL5);
|
||||
IREmit->_StoreMemPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, PReg, AddrNode);
|
||||
} else {
|
||||
// For X87 extended doubles, split before storing
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
auto DestAddr = IREmit->_Add(OpSize::i64Bit, AddrNode, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, DestAddr, Upper, OpSize::i64Bit);
|
||||
}
|
||||
} else {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, AddrNode, StackNode);
|
||||
}
|
||||
@@ -877,10 +885,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
if (ReducedPrecisionMode) {
|
||||
ResultNode = IREmit->_VFNeg(OpSize::i64Bit, OpSize::i64Bit, Value);
|
||||
} else {
|
||||
Ref Low = GetConstant(0);
|
||||
Ref High = GetConstant(0b1'000'0000'0000'0000ULL);
|
||||
Ref HelperNode = IREmit->_VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, Low);
|
||||
HelperNode = IREmit->_VInsGPR(OpSize::i128Bit, OpSize::i64Bit, 1, HelperNode, High);
|
||||
Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK);
|
||||
ResultNode = IREmit->_VXor(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
|
||||
}
|
||||
StoreStackValue(ResultNode);
|
||||
@@ -895,11 +900,8 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
ResultNode = IREmit->_VFAbs(OpSize::i64Bit, OpSize::i64Bit, Value);
|
||||
} else {
|
||||
// Intermediate insts
|
||||
Ref Low = GetConstant(~0ULL);
|
||||
Ref High = GetConstant(0b0'111'1111'1111'1111ULL);
|
||||
Ref HelperNode = IREmit->_VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, Low);
|
||||
HelperNode = IREmit->_VInsGPR(OpSize::i128Bit, OpSize::i64Bit, 1, HelperNode, High);
|
||||
ResultNode = IREmit->_VAnd(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
|
||||
Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK);
|
||||
ResultNode = IREmit->_VAndn(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
|
||||
}
|
||||
StoreStackValue(ResultNode);
|
||||
break;
|
||||
@@ -1025,7 +1027,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
return;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass() {
|
||||
return fextl::make_unique<X87StackOptimization>();
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures& Features) {
|
||||
return fextl::make_unique<X87StackOptimization>(Features);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -2118,7 +2118,7 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
LDUR |= Size << 30;
|
||||
LDUR |= AddrReg << 5;
|
||||
LDUR |= DataReg;
|
||||
LDUR |= Instr & (0b1'1111'1111 << 9);
|
||||
LDUR |= Instr & (0b1'1111'1111 << 12);
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
// Ordering matters with cross-thread visibility!
|
||||
std::atomic_ref<uint32_t>(PC[1]).store(DMB_LD, std::memory_order_release); // Back-patch the half-barrier.
|
||||
@@ -2132,7 +2132,7 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
STUR |= Size << 30;
|
||||
STUR |= AddrReg << 5;
|
||||
STUR |= DataReg;
|
||||
STUR |= Instr & (0b1'1111'1111 << 9);
|
||||
STUR |= Instr & (0b1'1111'1111 << 12);
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
std::atomic_ref<uint32_t>(PC[-1]).store(DMB, std::memory_order_release); // Back-patch the half-barrier.
|
||||
}
|
||||
|
||||
@@ -31,24 +31,28 @@ static bool LoadFileImpl(T& Data, const fextl::string& Filepath, size_t FixedSiz
|
||||
FileSize = FixedSize;
|
||||
}
|
||||
|
||||
ssize_t CurrentOffset = 0;
|
||||
ssize_t Read = -1;
|
||||
bool LoadedFile {};
|
||||
if (FileSize) {
|
||||
// File size is known upfront
|
||||
Data.resize(FileSize);
|
||||
Read = pread(FD, &Data.at(0), FileSize, 0);
|
||||
while (CurrentOffset != FileSize && (Read = pread(FD, &Data.at(CurrentOffset), FileSize, 0)) > 0) {
|
||||
CurrentOffset += Read;
|
||||
}
|
||||
|
||||
LoadedFile = Read == FileSize;
|
||||
LoadedFile = CurrentOffset == FileSize && Read != -1;
|
||||
} else {
|
||||
// The file is either empty or its size is unknown (e.g. procfs data).
|
||||
// Try reading in chunks instead
|
||||
ssize_t CurrentOffset = 0;
|
||||
constexpr size_t READ_SIZE = 4096;
|
||||
Data.resize(READ_SIZE);
|
||||
|
||||
while ((Read = pread(FD, &Data.at(CurrentOffset), READ_SIZE, CurrentOffset)) == READ_SIZE) {
|
||||
while ((Read = pread(FD, &Data.at(CurrentOffset), READ_SIZE, CurrentOffset)) > 0) {
|
||||
CurrentOffset += Read;
|
||||
Data.resize(CurrentOffset + Read);
|
||||
if ((CurrentOffset + READ_SIZE) > Data.size()) {
|
||||
Data.resize(CurrentOffset + READ_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
if (Read == -1) {
|
||||
|
||||
@@ -104,6 +104,9 @@ public:
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void HandleCallback(FEXCore::Core::InternalThreadState* Thread, uint64_t RIP) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual bool IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual bool IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
|
||||
///< State reconstruction helpers
|
||||
///< Reconstructs the guest RIP from the passed in thread context and related Host PC.
|
||||
FEX_DEFAULT_VISIBILITY virtual uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) = 0;
|
||||
@@ -121,7 +124,7 @@ public:
|
||||
* @return x86 EFLAGS reconstructed
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual uint32_t
|
||||
ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) = 0;
|
||||
ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs, uint64_t PSTATE) = 0;
|
||||
///< Sets FEX's internal EFLAGS representation to the passed in compacted form.
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) = 0;
|
||||
|
||||
|
||||
@@ -62,7 +62,7 @@ enum X86RegLocation : uint32_t {
|
||||
RFLAG_AF_RAW_LOC = 4, // Contains multiple bits, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_ZF_RAW_LOC = 6, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_SF_RAW_LOC = 7, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_TF_LOC = 8,
|
||||
RFLAG_TF_RAW_LOC = 8, // Contains multiple bits, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_IF_LOC = 9,
|
||||
RFLAG_DF_RAW_LOC = 10, // Contains multiple bits, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_OF_RAW_LOC = 11, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
|
||||
@@ -71,6 +71,16 @@ enum NamedVectorConstant : uint8_t {
|
||||
NAMED_VECTOR_X87_LOG10_2,
|
||||
NAMED_VECTOR_X87_LOG_2,
|
||||
|
||||
NAMED_VECTOR_CVTMAX_F32_I32,
|
||||
NAMED_VECTOR_CVTMAX_F32_I32_UPPER,
|
||||
NAMED_VECTOR_CVTMAX_F32_I64,
|
||||
NAMED_VECTOR_CVTMAX_F64_I32,
|
||||
NAMED_VECTOR_CVTMAX_F64_I32_UPPER,
|
||||
NAMED_VECTOR_CVTMAX_F64_I64,
|
||||
NAMED_VECTOR_CVTMAX_I32,
|
||||
NAMED_VECTOR_CVTMAX_I64,
|
||||
NAMED_VECTOR_F80_SIGN_MASK,
|
||||
|
||||
NAMED_VECTOR_CONST_POOL_MAX,
|
||||
// Beginning of named constants that don't have a constant pool backing.
|
||||
NAMED_VECTOR_ZERO = NAMED_VECTOR_CONST_POOL_MAX,
|
||||
|
||||
@@ -25,7 +25,7 @@ FMT_NODISCARD auto to_string(const fextl::fmt::basic_memory_buffer<Char, SIZE>&
|
||||
return fextl::basic_string<Char>(buf.data(), size);
|
||||
}
|
||||
|
||||
FMT_FUNC FMT_INLINE fextl::string vformat(::fmt::string_view fmt, ::fmt::format_args args) {
|
||||
FMT_INLINE fextl::string vformat(::fmt::string_view fmt, ::fmt::format_args args) {
|
||||
// Don't optimize the "{}" case to keep the binary size small and because it
|
||||
// can be better optimized in fmt::format anyway.
|
||||
auto buffer = memory_buffer();
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
# FEX - Fast x86 emulation frontend
|
||||
FEX allows you to run x86 and x86-64 binaries on an AArch64 host, similar to qemu-user and box86.
|
||||
It has native support for a rootfs overlay, so you don't need to chroot, as well as some thunklibs so it can forward things like GL to the host.
|
||||
FEX presents a Linux 5.0+ interface to the guest, and supports only AArch64 as a host.
|
||||
FEX presents a Linux 5.15+ interface to the guest, and supports only AArch64 as a host.
|
||||
FEX is very much work in progress, so expect things to change.
|
||||
|
||||
|
||||
|
||||
@@ -55,6 +55,9 @@ class HostFeatures(Flag) :
|
||||
FEATURE_CRYPTO = (1 << 10)
|
||||
FEATURE_AES256 = (1 << 11)
|
||||
FEATURE_SVEBITPERM = (1 << 12)
|
||||
FEATURE_TSO = (1 << 13)
|
||||
FEATURE_LRCPC = (1 << 14)
|
||||
FEATURE_LRCPC2 = (1 << 15)
|
||||
|
||||
HostFeaturesLookup = {
|
||||
"SVE128" : HostFeatures.FEATURE_SVE128,
|
||||
@@ -70,6 +73,9 @@ HostFeaturesLookup = {
|
||||
"CRYPTO" : HostFeatures.FEATURE_CRYPTO,
|
||||
"AES256" : HostFeatures.FEATURE_AES256,
|
||||
"SVEBITPERM" : HostFeatures.FEATURE_SVEBITPERM,
|
||||
"TSO" : HostFeatures.FEATURE_TSO,
|
||||
"LRCPC" : HostFeatures.FEATURE_LRCPC,
|
||||
"LRCPC2" : HostFeatures.FEATURE_LRCPC2,
|
||||
}
|
||||
|
||||
def GetHostFeatures(data):
|
||||
|
||||
@@ -506,6 +506,9 @@ int main(int argc, char** argv, char** const envp) {
|
||||
FEATURE_CRYPTO = (1U << 10),
|
||||
FEATURE_AES256 = (1U << 11),
|
||||
FEATURE_SVEBITPERM = (1U << 12),
|
||||
FEATURE_TSO = (1U << 13),
|
||||
FEATURE_LRCPC = (1U << 14),
|
||||
FEATURE_LRCPC2 = (1U << 15),
|
||||
};
|
||||
|
||||
uint64_t SVEWidth = 0;
|
||||
@@ -547,6 +550,20 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_SVEBITPERM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLESVEBITPERM);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_LRCPC) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLELRCPC);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_LRCPC2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLELRCPC2);
|
||||
}
|
||||
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_TSO) {
|
||||
// Always disable auto migration.
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOAUTOMIGRATION, "0");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "1");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, "1");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, "1");
|
||||
}
|
||||
|
||||
// Always enable ARMv8.1 LSE atomics.
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEATOMICS);
|
||||
@@ -584,6 +601,20 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_SVEBITPERM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLESVEBITPERM);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_LRCPC) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLELRCPC);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_LRCPC2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLELRCPC2);
|
||||
}
|
||||
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_TSO) {
|
||||
// Always disable auto migration.
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOAUTOMIGRATION, "0");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, "0");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, "0");
|
||||
}
|
||||
|
||||
// Always enable preserve_all abi.
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEPRESERVEALLABI);
|
||||
|
||||
@@ -64,8 +64,8 @@ $end_info$
|
||||
#include <sys/signal.h>
|
||||
|
||||
namespace {
|
||||
static bool SilentLog;
|
||||
static int OutputFD {-1};
|
||||
static bool SilentLog {};
|
||||
static int OutputFD {STDERR_FILENO};
|
||||
|
||||
void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
|
||||
if (SilentLog) {
|
||||
@@ -439,9 +439,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
|
||||
uint32_t KernelVersion = FEX::HLE::SyscallHandler::CalculateHostKernelVersion();
|
||||
if (KernelVersion < FEX::HLE::SyscallHandler::KernelVersion(4, 17)) {
|
||||
// We require 4.17 minimum for MAP_FIXED_NOREPLACE
|
||||
LogMan::Msg::EFmt("FEXLoader requires kernel 4.17 minimum. Expect problems.");
|
||||
if (KernelVersion < FEX::HLE::SyscallHandler::KernelVersion(5, 15)) {
|
||||
LogMan::Msg::EFmt("FEXLoader requires kernel 5.15 minimum. Expect problems.");
|
||||
}
|
||||
|
||||
// Before we go any further, set all of our host environment variables that the config has provided
|
||||
|
||||
@@ -77,13 +77,17 @@ fextl::string BuildOSXML() {
|
||||
return xml.str();
|
||||
}
|
||||
|
||||
fextl::string BuildTargetXML() {
|
||||
fextl::string BuildTargetXML(bool Is64Bit) {
|
||||
fextl::ostringstream xml;
|
||||
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
xml << "<!DOCTYPE target SYSTEM 'gdb-target.dtd'>\n";
|
||||
xml << "<target>\n";
|
||||
xml << "<architecture>i386:x86-64</architecture>\n";
|
||||
if (Is64Bit) {
|
||||
xml << "<architecture>i386:x86-64</architecture>\n";
|
||||
} else {
|
||||
xml << "<architecture>i386</architecture>\n";
|
||||
}
|
||||
xml << "<osabi>GNU/Linux</osabi>\n";
|
||||
xml << "<feature name='org.gnu.gdb.i386.core'>\n";
|
||||
|
||||
|
||||
@@ -47,5 +47,5 @@ fextl::string BuildOSXML();
|
||||
/**
|
||||
* @brief Returns the GDB specific construct of target describing XML.
|
||||
*/
|
||||
fextl::string BuildTargetXML();
|
||||
fextl::string BuildTargetXML(bool Is64Bit);
|
||||
} // namespace FEX::GDB::Info
|
||||
@@ -582,67 +582,17 @@ EmulatedFDManager::EmulatedFDManager(FEXCore::Context::Context* ctx)
|
||||
|
||||
EmulatedFDManager::~EmulatedFDManager() {}
|
||||
|
||||
int32_t EmulatedFDManager::OpenAt(int dirfs, const char* pathname, int flags, uint32_t mode) {
|
||||
char Tmp[PATH_MAX];
|
||||
const char* Path {};
|
||||
|
||||
int32_t EmulatedFDManager::Open(const char* pathname, int flags, uint32_t mode) {
|
||||
auto Creator = FDReadCreators.end();
|
||||
if (pathname) {
|
||||
Creator = FDReadCreators.find(pathname);
|
||||
Path = pathname;
|
||||
}
|
||||
|
||||
if (Creator == FDReadCreators.end()) {
|
||||
if (((pathname && pathname[0] != '/') || // If pathname exists then it must not be absolute
|
||||
!pathname) &&
|
||||
dirfs != AT_FDCWD) {
|
||||
// Passed in a dirfd that isn't magic FDCWD
|
||||
// We need to get the path from the fd now
|
||||
auto PathLength = FEX::get_fdpath(dirfs, Tmp);
|
||||
if (PathLength != -1) {
|
||||
if (pathname) {
|
||||
Tmp[PathLength] = '/';
|
||||
PathLength += 1;
|
||||
strncpy(&Tmp[PathLength], pathname, PATH_MAX - PathLength);
|
||||
} else {
|
||||
Tmp[PathLength] = '\0';
|
||||
}
|
||||
Path = Tmp;
|
||||
} else if (pathname) {
|
||||
Path = pathname;
|
||||
}
|
||||
} else {
|
||||
if (!pathname || pathname[0] == 0) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
Path = pathname;
|
||||
}
|
||||
|
||||
bool exists = access(Path, F_OK) == 0;
|
||||
bool RealPathExists = false;
|
||||
|
||||
if (exists) {
|
||||
// If realpath fails then the temporary buffer is in an undefined state.
|
||||
// Need to use another temporary just in-case realpath doesn't succeed.
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
char* RealPath = realpath(Path, ExistsTempPath);
|
||||
if (RealPath) {
|
||||
RealPathExists = true;
|
||||
Creator = FDReadCreators.find(RealPath);
|
||||
}
|
||||
}
|
||||
|
||||
if (!RealPathExists) {
|
||||
Creator = FDReadCreators.find(FHU::Filesystem::LexicallyNormal(Path));
|
||||
}
|
||||
|
||||
if (Creator == FDReadCreators.end()) {
|
||||
return -1;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
return Creator->second(CTX, dirfs, Path, flags, mode);
|
||||
return Creator->second(CTX, AT_FDCWD, pathname, flags, mode);
|
||||
}
|
||||
|
||||
int32_t EmulatedFDManager::ProcAuxv(FEXCore::Context::Context* ctx, int32_t fd, const char* pathname, int32_t flags, mode_t mode) {
|
||||
|
||||
@@ -23,7 +23,7 @@ class EmulatedFDManager {
|
||||
public:
|
||||
EmulatedFDManager(FEXCore::Context::Context* ctx);
|
||||
~EmulatedFDManager();
|
||||
int32_t OpenAt(int dirfs, const char* pathname, int flags, uint32_t mode);
|
||||
int32_t Open(const char* pathname, int flags, uint32_t mode);
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context* CTX;
|
||||
|
||||
@@ -29,6 +29,7 @@ $end_info$
|
||||
#include <algorithm>
|
||||
#include <errno.h>
|
||||
#include <cstring>
|
||||
#include <linux/openat2.h>
|
||||
#include <fcntl.h>
|
||||
#include <filesystem>
|
||||
#include <optional>
|
||||
@@ -339,6 +340,100 @@ FileManager::~FileManager() {
|
||||
close(RootFSFD);
|
||||
}
|
||||
|
||||
size_t FileManager::GetRootFSPrefixLen(const char* pathname, size_t len, bool AliasedOnly) {
|
||||
if (len < 2 || // If no pathname or root
|
||||
pathname[0] != '/') { // If we are getting root
|
||||
return 0;
|
||||
}
|
||||
|
||||
const auto& RootFSPath = LDPath();
|
||||
if (RootFSPath.empty()) { // If RootFS doesn't exist
|
||||
return 0;
|
||||
}
|
||||
|
||||
auto RootFSLen = RootFSPath.length();
|
||||
if (RootFSPath.ends_with("/")) {
|
||||
RootFSLen -= 1;
|
||||
}
|
||||
|
||||
if (RootFSLen > len) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (memcmp(pathname, RootFSPath.c_str(), RootFSLen) || (len > RootFSLen && pathname[RootFSLen] != '/')) {
|
||||
return 0; // If the path is not within the RootFS
|
||||
}
|
||||
|
||||
if (AliasedOnly) {
|
||||
fextl::string Path(pathname, len); // Need to nul-terminate so copy
|
||||
|
||||
struct stat HostStat {};
|
||||
struct stat RootFSStat {};
|
||||
if (lstat(Path.c_str(), &RootFSStat)) {
|
||||
LogMan::Msg::DFmt("GetRootFSPrefixLen: lstat on RootFS path failed: {}", std::string_view(pathname, len));
|
||||
return 0; // RootFS path does not exist?
|
||||
}
|
||||
if (lstat(Path.c_str() + RootFSLen, &HostStat)) {
|
||||
return 0; // Host path does not exist or not accessible
|
||||
}
|
||||
// Note: We do not check st_dev, since the RootFS might be
|
||||
// an overlayfs mount that changes it. This means there could
|
||||
// be false positives. However, since we check the size too,
|
||||
// this is highly unlikely (an overlaid file would need to
|
||||
// have the same exact size and coincidentally the same
|
||||
// inode number as on the host, which is implausible for things
|
||||
// like binaries and libraries).
|
||||
if (RootFSStat.st_size != HostStat.st_size || RootFSStat.st_ino != HostStat.st_ino || RootFSStat.st_mode != HostStat.st_mode) {
|
||||
return 0; // Host path is a different file
|
||||
}
|
||||
}
|
||||
|
||||
return RootFSLen;
|
||||
}
|
||||
|
||||
ssize_t FileManager::StripRootFSPrefix(char* pathname, ssize_t len, bool leaky) {
|
||||
if (len < 0) {
|
||||
return len;
|
||||
}
|
||||
|
||||
auto Prefix = GetRootFSPrefixLen(pathname, len, false);
|
||||
if (Prefix == 0) {
|
||||
return len;
|
||||
}
|
||||
|
||||
if (Prefix == len) {
|
||||
if (leaky) {
|
||||
// Getting the root, without a trailing /. This is a hack pressure-vessel uses to get the FEX RootFS,
|
||||
// so we have to leak it here...
|
||||
LogMan::Msg::DFmt("Leaking RootFS path for pressure-vessel");
|
||||
return len;
|
||||
} else {
|
||||
::strcpy(pathname, "/");
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
::memmove(pathname, pathname + Prefix, len - Prefix);
|
||||
pathname[len - Prefix] = '\0';
|
||||
|
||||
return len - Prefix;
|
||||
}
|
||||
|
||||
fextl::string FileManager::GetHostPath(fextl::string& Path, bool AliasedOnly) {
|
||||
auto Prefix = GetRootFSPrefixLen(Path.c_str(), Path.length(), AliasedOnly);
|
||||
|
||||
if (Prefix == 0) {
|
||||
return {};
|
||||
}
|
||||
|
||||
auto ret = Path.substr(Prefix);
|
||||
if (ret.empty()) { // Getting the root
|
||||
ret = "/";
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
fextl::string FileManager::GetEmulatedPath(const char* pathname, bool FollowSymlink) {
|
||||
if (!pathname || // If no pathname
|
||||
pathname[0] != '/' || // If relative
|
||||
@@ -438,8 +533,11 @@ std::pair<int, const char*> FileManager::GetEmulatedFDPath(int dirfd, const char
|
||||
// Get the symlink of RootFS FD + stripped subpath.
|
||||
auto SymlinkSize = FEX::HLE::GetSymlink(RootFSFD, &SubPath[1], CurrentTmp, PATH_MAX - 1);
|
||||
|
||||
if (SymlinkSize > 0 && CurrentTmp[0] == '/') {
|
||||
// If the symlink is absolute:
|
||||
// This might be a /proc symlink into the RootFS, so strip it in that case.
|
||||
SymlinkSize = StripRootFSPrefix(CurrentTmp, SymlinkSize, false);
|
||||
|
||||
if (SymlinkSize > 1 && CurrentTmp[0] == '/') {
|
||||
// If the symlink is absolute and not the root:
|
||||
// 1) Zero terminate it.
|
||||
// 2) Set the path as our current subpath.
|
||||
// 3) Switch to the next temporary index. (We don't want to overwrite the current one on the next loop iteration).
|
||||
@@ -515,23 +613,62 @@ static bool ShouldSkipOpenInEmu(int flags) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool FileManager::ReplaceEmuFd(int fd, int flags, uint32_t mode) {
|
||||
char Tmp[PATH_MAX + 1];
|
||||
|
||||
if (fd < 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Get the path of the file we just opened
|
||||
auto PathLength = FEX::get_fdpath(fd, Tmp);
|
||||
if (PathLength == -1) {
|
||||
return false;
|
||||
}
|
||||
Tmp[PathLength] = '\0';
|
||||
|
||||
// And try to open via EmuFD
|
||||
auto EmuFd = EmuFD.Open(Tmp, flags, mode);
|
||||
if (EmuFd == -1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// If we succeeded, swap out the fd
|
||||
::dup2(EmuFd, fd);
|
||||
::close(EmuFd);
|
||||
return true;
|
||||
}
|
||||
|
||||
uint64_t FileManager::Open(const char* pathname, int flags, uint32_t mode) {
|
||||
auto NewPath = GetSelf(pathname);
|
||||
const char* SelfPath = NewPath ? NewPath->data() : nullptr;
|
||||
int fd = -1;
|
||||
|
||||
if (!ShouldSkipOpenInEmu(flags)) {
|
||||
fd = EmuFD.OpenAt(AT_FDCWD, SelfPath, flags, mode);
|
||||
if (fd == -1) {
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(AT_FDCWD, SelfPath, true, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
fd = ::openat(Path.first, Path.second, flags, mode);
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(AT_FDCWD, SelfPath, false, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
FEX::HLE::open_how how = {
|
||||
.flags = (uint64_t)flags,
|
||||
.mode = (flags & (O_CREAT | O_TMPFILE)) ? mode & 07777 : 0, // openat2() is stricter about this
|
||||
.resolve = (Path.first == AT_FDCWD) ? 0u : RESOLVE_IN_ROOT, // AT_FDCWD means it's a thunk and not via RootFS
|
||||
};
|
||||
fd = ::syscall(SYSCALL_DEF(openat2), Path.first, Path.second, &how, sizeof(how));
|
||||
|
||||
if (fd == -1 && errno == EXDEV) {
|
||||
// This means a magic symlink (/proc/foo) was involved. In this case we
|
||||
// just punt and do the access without RESOLVE_IN_ROOT.
|
||||
fd = ::syscall(SYSCALL_DEF(openat), Path.first, Path.second, flags, mode);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (fd == -1) {
|
||||
// Open through RootFS failed (probably nonexistent), so open directly.
|
||||
if (fd == -1) {
|
||||
fd = ::open(SelfPath, flags, mode);
|
||||
}
|
||||
|
||||
ReplaceEmuFd(fd, flags, mode);
|
||||
} else {
|
||||
fd = ::open(SelfPath, flags, mode);
|
||||
}
|
||||
|
||||
@@ -658,20 +795,22 @@ uint64_t FileManager::Readlink(const char* pathname, char* buf, size_t bufsiz) {
|
||||
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(AT_FDCWD, pathname, false, TmpFilename);
|
||||
uint64_t Result = -1;
|
||||
if (Path.first != -1) {
|
||||
uint64_t Result = ::readlinkat(Path.first, Path.second, buf, bufsiz);
|
||||
if (Result != -1) {
|
||||
return Result;
|
||||
}
|
||||
Result = ::readlinkat(Path.first, Path.second, buf, bufsiz);
|
||||
|
||||
if (Result == -1 && errno == EINVAL) {
|
||||
// This means that the file wasn't a symlink
|
||||
// This is expected behaviour
|
||||
return -errno;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
if (Result == -1) {
|
||||
Result = ::readlink(pathname, buf, bufsiz);
|
||||
}
|
||||
|
||||
return ::readlink(pathname, buf, bufsiz);
|
||||
// We might have read a /proc/self/fd/* link. If so, strip the RootFS prefix from it.
|
||||
return StripRootFSPrefix(buf, Result, true);
|
||||
}
|
||||
|
||||
uint64_t FileManager::Chmod(const char* pathname, mode_t mode) {
|
||||
@@ -733,20 +872,24 @@ uint64_t FileManager::Readlinkat(int dirfd, const char* pathname, char* buf, siz
|
||||
|
||||
FDPathTmpData TmpFilename;
|
||||
auto NewPath = GetEmulatedFDPath(dirfd, pathname, false, TmpFilename);
|
||||
uint64_t Result = -1;
|
||||
|
||||
if (NewPath.first != -1) {
|
||||
uint64_t Result = ::readlinkat(NewPath.first, NewPath.second, buf, bufsiz);
|
||||
if (Result != -1) {
|
||||
return Result;
|
||||
}
|
||||
Result = ::readlinkat(NewPath.first, NewPath.second, buf, bufsiz);
|
||||
|
||||
if (Result == -1 && errno == EINVAL) {
|
||||
// This means that the file wasn't a symlink
|
||||
// This is expected behaviour
|
||||
return -errno;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
return ::readlinkat(dirfd, pathname, buf, bufsiz);
|
||||
if (Result == -1) {
|
||||
Result = ::readlinkat(dirfd, pathname, buf, bufsiz);
|
||||
}
|
||||
|
||||
// We might have read a /proc/self/fd/* link. If so, strip the RootFS prefix from it.
|
||||
return StripRootFSPrefix(buf, Result, true);
|
||||
}
|
||||
|
||||
uint64_t FileManager::Openat([[maybe_unused]] int dirfs, const char* pathname, int flags, uint32_t mode) {
|
||||
@@ -756,17 +899,29 @@ uint64_t FileManager::Openat([[maybe_unused]] int dirfs, const char* pathname, i
|
||||
int32_t fd = -1;
|
||||
|
||||
if (!ShouldSkipOpenInEmu(flags)) {
|
||||
fd = EmuFD.OpenAt(dirfs, SelfPath, flags, mode);
|
||||
if (fd == -1) {
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(dirfs, SelfPath, true, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(dirfs, SelfPath, false, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
FEX::HLE::open_how how = {
|
||||
.flags = (uint64_t)flags,
|
||||
.mode = (flags & (O_CREAT | O_TMPFILE)) ? mode & 07777 : 0, // openat2() is stricter about this,
|
||||
.resolve = (Path.first == AT_FDCWD) ? 0u : RESOLVE_IN_ROOT, // AT_FDCWD means it's a thunk and not via RootFS
|
||||
};
|
||||
fd = ::syscall(SYSCALL_DEF(openat2), Path.first, Path.second, &how, sizeof(how));
|
||||
if (fd == -1 && errno == EXDEV) {
|
||||
// This means a magic symlink (/proc/foo) was involved. In this case we
|
||||
// just punt and do the access without RESOLVE_IN_ROOT.
|
||||
fd = ::syscall(SYSCALL_DEF(openat), Path.first, Path.second, flags, mode);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (fd == -1) {
|
||||
// Open through RootFS failed (probably nonexistent), so open directly.
|
||||
if (fd == -1) {
|
||||
fd = ::syscall(SYSCALL_DEF(openat), dirfs, SelfPath, flags, mode);
|
||||
}
|
||||
|
||||
ReplaceEmuFd(fd, flags, mode);
|
||||
} else {
|
||||
fd = ::syscall(SYSCALL_DEF(openat), dirfs, SelfPath, flags, mode);
|
||||
}
|
||||
|
||||
@@ -780,17 +935,29 @@ uint64_t FileManager::Openat2(int dirfs, const char* pathname, FEX::HLE::open_ho
|
||||
int32_t fd = -1;
|
||||
|
||||
if (!ShouldSkipOpenInEmu(how->flags)) {
|
||||
fd = EmuFD.OpenAt(dirfs, SelfPath, how->flags, how->mode);
|
||||
if (fd == -1) {
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(dirfs, SelfPath, true, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(dirfs, SelfPath, false, TmpFilename);
|
||||
if (Path.first != -1 && !(how->resolve & RESOLVE_IN_ROOT)) {
|
||||
// AT_FDCWD means it's a thunk and not via RootFS
|
||||
if (Path.first != AT_FDCWD) {
|
||||
how->resolve |= RESOLVE_IN_ROOT;
|
||||
}
|
||||
fd = ::syscall(SYSCALL_DEF(openat2), Path.first, Path.second, how, usize);
|
||||
how->resolve &= ~RESOLVE_IN_ROOT;
|
||||
if (fd == -1 && errno == EXDEV) {
|
||||
// This means a magic symlink (/proc/foo) was involved. In this case we
|
||||
// just punt and do the access without RESOLVE_IN_ROOT.
|
||||
fd = ::syscall(SYSCALL_DEF(openat2), Path.first, Path.second, how, usize);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (fd == -1) {
|
||||
// Open through RootFS failed (probably nonexistent), so open directly.
|
||||
if (fd == -1) {
|
||||
fd = ::syscall(SYSCALL_DEF(openat2), dirfs, SelfPath, how, usize);
|
||||
}
|
||||
|
||||
ReplaceEmuFd(fd, how->flags, how->mode);
|
||||
} else {
|
||||
fd = ::syscall(SYSCALL_DEF(openat2), dirfs, SelfPath, how, usize);
|
||||
}
|
||||
|
||||
|
||||
@@ -85,9 +85,12 @@ public:
|
||||
bool IsRootFSFD(int dirfd, uint64_t inode);
|
||||
|
||||
fextl::string GetEmulatedPath(const char* pathname, bool FollowSymlink = false);
|
||||
fextl::string GetHostPath(fextl::string& Path, bool AliasedOnly);
|
||||
using FDPathTmpData = std::array<char[PATH_MAX], 2>;
|
||||
std::pair<int, const char*> GetEmulatedFDPath(int dirfd, const char* pathname, bool FollowSymlink, FDPathTmpData& TmpFilename);
|
||||
|
||||
bool ReplaceEmuFd(int fd, int flags, uint32_t mode);
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
void TrackFEXFD(int FD) noexcept {
|
||||
std::lock_guard lk(FEXTrackingFDMutex);
|
||||
@@ -140,6 +143,8 @@ private:
|
||||
#endif
|
||||
|
||||
bool RootFSPathExists(const char* Filepath);
|
||||
size_t GetRootFSPrefixLen(const char* pathname, size_t len, bool AliasedOnly);
|
||||
ssize_t StripRootFSPrefix(char* pathname, ssize_t len, bool leaky);
|
||||
|
||||
struct ThunkDBObject {
|
||||
fextl::string LibraryName;
|
||||
|
||||
@@ -92,7 +92,8 @@ GdbServer::~GdbServer() {
|
||||
|
||||
GdbServer::GdbServer(FEXCore::Context::Context* ctx, FEX::HLE::SignalDelegator* SignalDelegation, FEX::HLE::SyscallHandler* const SyscallHandler)
|
||||
: CTX(ctx)
|
||||
, SyscallHandler {SyscallHandler} {
|
||||
, SyscallHandler {SyscallHandler}
|
||||
, SignalDelegation {SignalDelegation} {
|
||||
// Pass all signals by default
|
||||
std::fill(PassSignals.begin(), PassSignals.end(), true);
|
||||
|
||||
@@ -107,10 +108,21 @@ GdbServer::GdbServer(FEXCore::Context::Context* ctx, FEX::HLE::SignalDelegator*
|
||||
return false;
|
||||
}
|
||||
|
||||
auto ThreadObject = FEX::HLE::ThreadManager::GetStateObjectFromFEXCoreThread(Thread);
|
||||
ThreadObject->GdbInfo = {};
|
||||
ThreadObject->GdbInfo->Signal = Signal;
|
||||
|
||||
ThreadObject->GdbInfo->SignalPC = ArchHelpers::Context::GetPc(ucontext);
|
||||
this->SignalDelegation->SpillSRA(Thread, ucontext, Thread->CurrentFrame->InSyscallInfo);
|
||||
|
||||
memcpy(ThreadObject->GdbInfo->GPRs, ArchHelpers::Context::GetArmGPRs(ucontext), sizeof(ThreadObject->GdbInfo->GPRs));
|
||||
ThreadObject->GdbInfo->PState = ArchHelpers::Context::GetArmPState(ucontext);
|
||||
|
||||
// Let GDB know that we have a signal
|
||||
this->Break(Thread, Signal);
|
||||
|
||||
WaitForThreadWakeup();
|
||||
ThreadObject->GdbInfo.reset();
|
||||
|
||||
return true;
|
||||
},
|
||||
@@ -145,6 +157,10 @@ static fextl::string hexstring(fextl::istringstream& ss, int delm) {
|
||||
return ret;
|
||||
}
|
||||
|
||||
static fextl::string appendHex(const char* data, size_t length) {
|
||||
return fextl::fmt::format("{:02x}", fmt::join(data, data + length, ""));
|
||||
}
|
||||
|
||||
static fextl::string encodeHex(const unsigned char* data, size_t length) {
|
||||
fextl::ostringstream ss;
|
||||
|
||||
@@ -271,7 +287,6 @@ const FEX::HLE::ThreadStateObject* GdbServer::FindThreadByTID(uint32_t TID) {
|
||||
return Threads->at(0);
|
||||
}
|
||||
|
||||
|
||||
GdbServer::GDBContextDefinition GdbServer::GenerateContextDefinition(const FEX::HLE::ThreadStateObject* ThreadObject) {
|
||||
GDBContextDefinition GDB {};
|
||||
FEXCore::Core::CPUState state {};
|
||||
@@ -281,9 +296,16 @@ GdbServer::GDBContextDefinition GdbServer::GenerateContextDefinition(const FEX::
|
||||
|
||||
// Encode the GDB context definition
|
||||
memcpy(&GDB.gregs[0], &state.gregs[0], sizeof(GDB.gregs));
|
||||
memcpy(&GDB.rip, &state.rip, sizeof(GDB.rip));
|
||||
if (ThreadObject->GdbInfo.has_value()) {
|
||||
GDB.rip = CTX->RestoreRIPFromHostPC(ThreadObject->Thread, ThreadObject->GdbInfo->SignalPC);
|
||||
|
||||
GDB.eflags = CTX->ReconstructCompactedEFLAGS(ThreadObject->Thread, false, nullptr, 0);
|
||||
const bool WasInJIT = CTX->IsAddressInCodeBuffer(ThreadObject->Thread, ThreadObject->GdbInfo->SignalPC);
|
||||
GDB.eflags = CTX->ReconstructCompactedEFLAGS(ThreadObject->Thread, WasInJIT, const_cast<uint64_t*>(ThreadObject->GdbInfo->GPRs),
|
||||
ThreadObject->GdbInfo->PState);
|
||||
} else {
|
||||
GDB.rip = ThreadObject->Thread->CurrentFrame->State.rip;
|
||||
GDB.eflags = CTX->ReconstructCompactedEFLAGS(ThreadObject->Thread, false, nullptr, 0);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
memcpy(&GDB.mm[i], &state.mm[i], sizeof(GDB.mm[i]));
|
||||
@@ -302,8 +324,8 @@ GdbServer::GDBContextDefinition GdbServer::GenerateContextDefinition(const FEX::
|
||||
|
||||
CTX->ReconstructXMMRegisters(ThreadObject->Thread, XMM_Low, YMM_High);
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
memcpy(&GDB.xmm[0], &XMM_Low[i], sizeof(__uint128_t));
|
||||
memcpy(&GDB.xmm[2], &YMM_High[i], sizeof(__uint128_t));
|
||||
memcpy(&GDB.xmm[i][0], &XMM_Low[i], sizeof(__uint128_t));
|
||||
memcpy(&GDB.xmm[i][2], &YMM_High[i], sizeof(__uint128_t));
|
||||
}
|
||||
|
||||
return GDB;
|
||||
@@ -406,7 +428,7 @@ GdbServer::HandledPacketType GdbServer::XferCommandExecFile(const fextl::string&
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::XferCommandFeatures(const fextl::string& annex, int offset, int length) {
|
||||
if (annex == "target.xml") {
|
||||
return {EncodeXferString(GDB::Info::BuildTargetXML(), offset, length), HandledPacketType::TYPE_ACK};
|
||||
return {EncodeXferString(GDB::Info::BuildTargetXML(Is64BitMode()), offset, length), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
@@ -639,8 +661,39 @@ GdbServer::HandledPacketType GdbServer::CommandReadRegisters(const fextl::string
|
||||
// Pause up front
|
||||
SyscallHandler->TM.Pause();
|
||||
const FEX::HLE::ThreadStateObject* CurrentThread = FindThreadByTID(CurrentDebuggingThread);
|
||||
const size_t NumGPR = Is64BitMode() ? FEXCore::Core::CPUState::NUM_GPRS : FEXCore::Core::CPUState::NUM_GPRS / 2;
|
||||
const size_t GPRSize = Is64BitMode() ? sizeof(uint64_t) : sizeof(uint32_t);
|
||||
const size_t NumXMM = Is64BitMode() ? FEXCore::Core::CPUState::NUM_XMMS : FEXCore::Core::CPUState::NUM_XMMS / 2;
|
||||
const size_t XMMSize = Is64BitMode() ? sizeof(__uint128_t) * 2 : sizeof(__uint128_t);
|
||||
fextl::string str;
|
||||
auto GDB = GenerateContextDefinition(CurrentThread);
|
||||
return {encodeHex((unsigned char*)&GDB, sizeof(GDBContextDefinition)), HandledPacketType::TYPE_ACK};
|
||||
for (size_t i = 0; i < NumGPR; ++i) {
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.gregs[i]), GPRSize);
|
||||
}
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.rip), GPRSize);
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.eflags), sizeof(uint32_t));
|
||||
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.cs), sizeof(uint32_t));
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.ss), sizeof(uint32_t));
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.ds), sizeof(uint32_t));
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.es), sizeof(uint32_t));
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.fs), sizeof(uint32_t));
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.gs), sizeof(uint32_t));
|
||||
for (auto& mm : GDB.mm) {
|
||||
str += appendHex(reinterpret_cast<const char*>(&mm), sizeof(X80Float));
|
||||
}
|
||||
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.fctrl), sizeof(uint32_t));
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.fstat), sizeof(uint32_t));
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.dummies), sizeof(GDB.dummies));
|
||||
|
||||
for (size_t i = 0; i < NumXMM; ++i) {
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.xmm[i]), XMMSize);
|
||||
}
|
||||
|
||||
str += appendHex(reinterpret_cast<const char*>(&GDB.mxcsr), sizeof(uint32_t));
|
||||
|
||||
return {str, HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::CommandThreadOp(const fextl::string& packet) {
|
||||
@@ -1118,7 +1171,10 @@ GdbServer::HandledPacketType GdbServer::CommandMultiLetterV(const fextl::string&
|
||||
return HandlevFile(packet);
|
||||
}
|
||||
|
||||
// TODO: vKill
|
||||
if (packet.starts_with("vKill")) {
|
||||
tgkill(::getpid(), ::getpid(), SIGKILL);
|
||||
}
|
||||
|
||||
// TODO: vRun
|
||||
// TODO: vStopped
|
||||
|
||||
|
||||
@@ -145,6 +145,7 @@ private:
|
||||
|
||||
FEXCore::Context::Context* CTX;
|
||||
FEX::HLE::SyscallHandler* const SyscallHandler;
|
||||
FEX::HLE::SignalDelegator* SignalDelegation;
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
fextl::unique_ptr<std::iostream> CommsStream;
|
||||
std::mutex sendMutex;
|
||||
|
||||
@@ -372,6 +372,7 @@ bool SignalDelegator::HandleDispatcherGuestSignal(FEXCore::Core::InternalThreadS
|
||||
|
||||
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.sigaction);
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
Frame->State.flags[FEXCore::X86State::RFLAG_TF_RAW_LOC] = 0;
|
||||
|
||||
// The guest starts its signal frame with a zero initialized FPU
|
||||
// Set that up now. Little bit costly but it's a requirement
|
||||
@@ -577,18 +578,8 @@ void SignalDelegator::HandleGuestSignal(FEX::HLE::ThreadStateObject* ThreadObjec
|
||||
ERROR_AND_DIE_FMT("X86 shouldn't hit this InterruptFaultPage");
|
||||
#endif
|
||||
}
|
||||
} else if (Signal == SIGSEGV && (SigInfo.si_code == SEGV_MAPERR || SigInfo.si_code == SEGV_ACCERR) &&
|
||||
FaultSafeUserMemAccess::IsFaultLocation(ArchHelpers::Context::GetPc(UContext))) {
|
||||
// If you want to emulate EFAULT behaviour then enable this if-statement.
|
||||
// Do this once we find an application that depends on this.
|
||||
if constexpr (false) {
|
||||
// Return from the subroutine, returning EFAULT.
|
||||
ArchHelpers::Context::SetArmReg(UContext, 0, EFAULT);
|
||||
ArchHelpers::Context::SetPc(UContext, ArchHelpers::Context::GetArmReg(UContext, 30));
|
||||
return;
|
||||
} else {
|
||||
ERROR_AND_DIE_FMT("Received invalid data to syscall. Crashing now!");
|
||||
}
|
||||
} else if (FaultSafeUserMemAccess::TryHandleSafeFault(Signal, SigInfo, UContext)) {
|
||||
ERROR_AND_DIE_FMT("Received invalid data to syscall. Crashing now!");
|
||||
} else {
|
||||
if (IsAsyncSignal(&SigInfo, Signal) && MustDeferSignal) {
|
||||
// If the signal is asynchronous (as determined by si_code) and FEX is in a state of needing
|
||||
|
||||
@@ -137,6 +137,9 @@ public:
|
||||
}
|
||||
|
||||
void SaveTelemetry();
|
||||
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState* Thread, void* ucontext, uint32_t IgnoreMask);
|
||||
|
||||
private:
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleGuestSignal(FEX::HLE::ThreadStateObject* ThreadObject, int Signal, void* Info, void* UContext);
|
||||
@@ -242,8 +245,6 @@ private:
|
||||
///< FP state now follows after this.
|
||||
};
|
||||
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState* Thread, void* ucontext, uint32_t IgnoreMask);
|
||||
|
||||
void RestoreFrame_x64(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* Context,
|
||||
FEXCore::Core::CpuStateFrame* Frame, void* ucontext);
|
||||
void RestoreFrame_ia32(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* Context,
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
|
||||
#include "CodeLoader.h"
|
||||
|
||||
#include "FEXHeaderUtils/StringArgumentParser.h"
|
||||
#include "Linux/Utils/ELFContainer.h"
|
||||
#include "Linux/Utils/ELFParser.h"
|
||||
|
||||
@@ -139,30 +140,20 @@ template uint64_t GetDentsEmulation<false>(int, FEX::HLE::x64::linux_dirent*, ui
|
||||
|
||||
template uint64_t GetDentsEmulation<true>(int, FEX::HLE::x32::linux_dirent_32*, uint32_t);
|
||||
|
||||
static bool IsShebangFile(std::span<char> Data) {
|
||||
static fextl::string GetShebangInterpFile(std::span<char> Data) {
|
||||
// File isn't large enough to even contain a shebang.
|
||||
if (Data.size() <= 2) {
|
||||
return false;
|
||||
return {};
|
||||
}
|
||||
|
||||
// Handle shebang files.
|
||||
if (Data[0] == '#' && Data[1] == '!') {
|
||||
fextl::string InterpreterLine {Data.begin() + 2, // strip off "#!" prefix
|
||||
std::find(Data.begin(), Data.end(), '\n')};
|
||||
fextl::vector<fextl::string> ShebangArguments {};
|
||||
|
||||
// Shebang line can have a single argument
|
||||
fextl::istringstream InterpreterSS(InterpreterLine);
|
||||
fextl::string Argument;
|
||||
while (std::getline(InterpreterSS, Argument, ' ')) {
|
||||
if (Argument.empty()) {
|
||||
continue;
|
||||
}
|
||||
ShebangArguments.push_back(std::move(Argument));
|
||||
}
|
||||
fextl::vector<std::string_view> ShebangArguments = FHU::ParseArgumentsFromString(InterpreterLine);
|
||||
|
||||
// Executable argument
|
||||
fextl::string& ShebangProgram = ShebangArguments[0];
|
||||
fextl::string ShebangProgram(ShebangArguments[0]);
|
||||
|
||||
// If the filename is absolute then prepend the rootfs
|
||||
// If it is relative then don't append the rootfs
|
||||
@@ -170,13 +161,15 @@ static bool IsShebangFile(std::span<char> Data) {
|
||||
ShebangProgram = FEX::HLE::_SyscallHandler->RootFSPath() + ShebangProgram;
|
||||
}
|
||||
|
||||
return FHU::Filesystem::Exists(ShebangProgram);
|
||||
if (FHU::Filesystem::Exists(ShebangProgram)) {
|
||||
return ShebangProgram;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
return {};
|
||||
}
|
||||
|
||||
static bool IsShebangFD(int FD) {
|
||||
static fextl::string GetShebangInterpFD(int FD) {
|
||||
// We don't know the state of the FD coming in since this might be a guest tracked FD.
|
||||
// Need to be extra careful here not to adjust file offsets and status flags.
|
||||
//
|
||||
@@ -187,19 +180,19 @@ static bool IsShebangFD(int FD) {
|
||||
const auto ChunkSize = 257l;
|
||||
const auto ReadSize = pread(FD, &Header.at(0), ChunkSize, 0);
|
||||
|
||||
return IsShebangFile(std::span<char>(Header.data(), ReadSize));
|
||||
return GetShebangInterpFile(std::span<char>(Header.data(), ReadSize));
|
||||
}
|
||||
|
||||
static bool IsShebangFilename(const fextl::string& Filename) {
|
||||
static fextl::string GetShebangInterpFilename(const fextl::string& Filename) {
|
||||
// Open the Filename to determine if it is a shebang file.
|
||||
int FD = open(Filename.c_str(), O_RDONLY | O_CLOEXEC);
|
||||
if (FD == -1) {
|
||||
return false;
|
||||
return {};
|
||||
}
|
||||
|
||||
bool IsShebang = IsShebangFD(FD);
|
||||
auto Interp = GetShebangInterpFD(FD);
|
||||
close(FD);
|
||||
return IsShebang;
|
||||
return Interp;
|
||||
}
|
||||
|
||||
uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname, char* const* argv, char* const* envp, ExecveAtArgs Args) {
|
||||
@@ -208,18 +201,19 @@ uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname
|
||||
|
||||
fextl::string RootFS = SyscallHandler->RootFSPath();
|
||||
ELFLoader::ELFContainer::ELFType Type {};
|
||||
ELFLoader::ELFContainer::ELFType InterpreterType {};
|
||||
|
||||
// AT_EMPTY_PATH is only used if the pathname is empty.
|
||||
const bool IsFDExec = (Args.flags & AT_EMPTY_PATH) && strlen(pathname) == 0;
|
||||
fextl::string FDExecEnv;
|
||||
fextl::string FDSeccompEnv;
|
||||
|
||||
bool IsShebang {};
|
||||
fextl::string ShebangInterpreter {};
|
||||
|
||||
if (IsFDExec) {
|
||||
Type = ELFLoader::ELFContainer::GetELFType(Args.dirfd);
|
||||
|
||||
IsShebang = IsShebangFD(Args.dirfd);
|
||||
ShebangInterpreter = GetShebangInterpFD(Args.dirfd);
|
||||
} else {
|
||||
// For absolute paths, check the rootfs first (if available)
|
||||
if (pathname[0] == '/') {
|
||||
@@ -253,7 +247,12 @@ uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname
|
||||
|
||||
Type = ELFLoader::ELFContainer::GetELFType(Filename);
|
||||
|
||||
IsShebang = IsShebangFilename(Filename);
|
||||
ShebangInterpreter = GetShebangInterpFilename(Filename);
|
||||
}
|
||||
|
||||
const bool IsShebang = !ShebangInterpreter.empty();
|
||||
if (IsShebang) {
|
||||
InterpreterType = ELFLoader::ELFContainer::GetELFType(ShebangInterpreter);
|
||||
}
|
||||
|
||||
if (!IsShebang && Type == ELFLoader::ELFContainer::ELFType::TYPE_NONE) {
|
||||
@@ -306,6 +305,10 @@ uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname
|
||||
// - FEXServer FD inheritance (unshare(CLONE_NEWNET))
|
||||
const bool NeedsEnvpCopy = (IsFDExec && !(IsBinfmtCompatible || IsOtherELF)) || HasSeccomp;
|
||||
|
||||
// We are trying to execute a shebang handled by a different architecture interpreter (e.g. /usr/bin/python from the host FS).
|
||||
// In this case we just defer to the kernel.
|
||||
const bool IsForeignShebang = (IsShebang && InterpreterType == ELFLoader::ELFContainer::ELFType::TYPE_OTHER_ELF);
|
||||
|
||||
if (NeedsEnvpCopy) {
|
||||
if (envp) {
|
||||
auto OldEnvp = envp;
|
||||
@@ -354,13 +357,36 @@ uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname
|
||||
EnvpPtr = const_cast<char* const*>(EnvpArgs.data());
|
||||
}
|
||||
|
||||
if (IsBinfmtCompatible || IsOtherELF) {
|
||||
if (!IsFDExec && (IsForeignShebang || IsOtherELF || !IsBinfmtCompatible)) {
|
||||
// With a merged RootFS, the entire real filesystem is visible through the rootfs
|
||||
// prefix. If we are executing a non-emulated binary, we should do so through the host
|
||||
// path.
|
||||
|
||||
auto Path = SyscallHandler->FM.GetHostPath(Filename, true);
|
||||
if (!Path.empty() && FHU::Filesystem::Exists(Path)) {
|
||||
Filename = std::move(Path);
|
||||
}
|
||||
}
|
||||
|
||||
if (IsBinfmtCompatible || IsOtherELF || IsForeignShebang) {
|
||||
Result = ::syscall(SYS_execveat, Args.dirfd, Filename.c_str(), argv, EnvpPtr, Args.flags);
|
||||
CloseSeccompFD();
|
||||
CloseFDExecFD();
|
||||
SYSCALL_ERRNO();
|
||||
}
|
||||
|
||||
// If we are executing an emulated interpreter shebang file through the loader,
|
||||
// we need to strip the RootFS prefix. The loader will pass this filename to the
|
||||
// interpreter as-is, which will access it using RootFS redirection.
|
||||
// Note that unlike above, the prefix is stripped unconditionally (AliasedOnly=false),
|
||||
// and the script path need not exist in the host.
|
||||
if (IsShebang) {
|
||||
auto Path = SyscallHandler->FM.GetHostPath(Filename, false);
|
||||
if (!Path.empty()) {
|
||||
Filename = std::move(Path);
|
||||
}
|
||||
}
|
||||
|
||||
// We don't have an interpreter installed or we are executing a non-ELF executable
|
||||
// We now need to munge the arguments
|
||||
fextl::vector<const char*> ExecveArgs {};
|
||||
@@ -500,7 +526,7 @@ static uint64_t Clone2Handler(FEXCore::Core::CpuStateFrame* Frame, FEX::HLE::clo
|
||||
|
||||
// Remove flags that will break us
|
||||
constexpr uint64_t INVALID_FOR_HOST = CLONE_SETTLS;
|
||||
uint64_t Flags = args->args.flags & ~INVALID_FOR_HOST;
|
||||
uint64_t Flags = (args->args.flags & ~INVALID_FOR_HOST) | args->args.exit_signal;
|
||||
uint64_t Result = ::clone(Clone2HandlerRet, // To be called function
|
||||
(void*)((uint64_t)args->NewStack + args->StackSize), // Stack
|
||||
Flags, // Flags
|
||||
@@ -650,9 +676,7 @@ uint64_t CloneHandler(FEXCore::Core::CpuStateFrame* Frame, FEX::HLE::clone3_args
|
||||
|
||||
if (!(flags & CLONE_THREAD)) {
|
||||
// CLONE_PARENT is ignored (Implied by CLONE_THREAD)
|
||||
return FEX::HLE::ForkGuest(Thread, Frame, flags, reinterpret_cast<void*>(args->args.stack), args->args.stack_size,
|
||||
reinterpret_cast<pid_t*>(args->args.parent_tid), reinterpret_cast<pid_t*>(args->args.child_tid),
|
||||
reinterpret_cast<void*>(args->args.tls));
|
||||
return FEX::HLE::ForkGuest(Thread, Frame, args);
|
||||
} else {
|
||||
auto NewThread = FEX::HLE::CreateNewThread(Thread->CTX, Frame, args);
|
||||
|
||||
@@ -782,8 +806,8 @@ uint32_t SyscallHandler::CalculateHostKernelVersion() {
|
||||
}
|
||||
|
||||
uint32_t SyscallHandler::CalculateGuestKernelVersion() {
|
||||
// We currently only emulate a kernel between the ranges of Kernel 5.0.0 and 6.11.0
|
||||
return std::max(KernelVersion(5, 0), std::min(KernelVersion(6, 11), GetHostKernelVersion()));
|
||||
// We currently only emulate a kernel between the ranges of Kernel 5.15.0 and 6.11.0
|
||||
return std::max(KernelVersion(5, 15), std::min(KernelVersion(6, 11), GetHostKernelVersion()));
|
||||
}
|
||||
|
||||
uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) {
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "LinuxSyscalls/LinuxAllocator.h"
|
||||
#include "LinuxSyscalls/ThreadManager.h"
|
||||
#include "LinuxSyscalls/Seccomp/SeccompEmulator.h"
|
||||
#include "ArchHelpers/MContext.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
@@ -674,6 +675,18 @@ namespace FaultSafeUserMemAccess {
|
||||
}
|
||||
#endif
|
||||
bool IsFaultLocation(uint64_t PC);
|
||||
|
||||
static inline bool TryHandleSafeFault(int Signal, const siginfo_t& SigInfo, void* UContext) {
|
||||
if (Signal == SIGSEGV && (SigInfo.si_code == SEGV_MAPERR || SigInfo.si_code == SEGV_ACCERR) &&
|
||||
FaultSafeUserMemAccess::IsFaultLocation(ArchHelpers::Context::GetPc(UContext))) {
|
||||
// Return from the subroutine, returning EFAULT.
|
||||
ArchHelpers::Context::SetArmReg(UContext, 0, EFAULT);
|
||||
ArchHelpers::Context::SetPc(UContext, ArchHelpers::Context::GetArmReg(UContext, 30));
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
} // namespace FaultSafeUserMemAccess
|
||||
|
||||
} // namespace FEX::HLE
|
||||
|
||||
@@ -110,29 +110,22 @@ void RegisterFD(FEX::HLE::SyscallHandler* Handler) {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 8, 0)) {
|
||||
// Only exists on kernel 5.8+
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(faccessat2, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int dirfd, const char* pathname, int mode, int flags) -> uint64_t {
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.FAccessat2(dirfd, pathname, mode, flags);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(faccessat2, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int dirfd, const char* pathname, int mode, int flags) -> uint64_t {
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.FAccessat2(dirfd, pathname, mode, flags);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(
|
||||
openat2, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int dirfs, const char* pathname, struct open_how* how, size_t usize) -> uint64_t {
|
||||
open_how HostHow {};
|
||||
size_t HostSize = std::min(sizeof(open_how), usize);
|
||||
memcpy(&HostHow, how, HostSize);
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(openat2, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int dirfs, const char* pathname, struct open_how* how, size_t usize) -> uint64_t {
|
||||
open_how HostHow {};
|
||||
size_t HostSize = std::min(sizeof(open_how), usize);
|
||||
memcpy(&HostHow, how, HostSize);
|
||||
|
||||
HostHow.flags = FEX::HLE::RemapFromX86Flags(HostHow.flags);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.Openat2(dirfs, pathname, &HostHow, HostSize);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(faccessat2, UnimplementedSyscallSafe);
|
||||
REGISTER_SYSCALL_IMPL(openat2, UnimplementedSyscallSafe);
|
||||
}
|
||||
HostHow.flags = FEX::HLE::RemapFromX86Flags(HostHow.flags);
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.Openat2(dirfs, pathname, &HostHow, HostSize);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(eventfd, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, uint32_t count) -> uint64_t {
|
||||
@@ -155,14 +148,10 @@ void RegisterFD(FEX::HLE::SyscallHandler* Handler) {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 9, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(close_range, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, unsigned int first, unsigned int last, unsigned int flags) -> uint64_t {
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.CloseRange(first, last, flags);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(close_range, UnimplementedSyscallSafe);
|
||||
}
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(close_range, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, unsigned int first, unsigned int last, unsigned int flags) -> uint64_t {
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.CloseRange(first, last, flags);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
}
|
||||
} // namespace FEX::HLE
|
||||
@@ -492,90 +492,42 @@ void RegisterCommon(FEX::HLE::SyscallHandler* Handler) {
|
||||
SyscallPassthrough2<SYSCALL_DEF(pkey_alloc)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(pkey_free, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough1<SYSCALL_DEF(pkey_free)>);
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 1, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(io_uring_setup, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(io_uring_setup)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(io_uring_enter, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough6<SYSCALL_DEF(io_uring_enter)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(io_uring_register, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(io_uring_register)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(open_tree, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(open_tree)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(move_mount, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(move_mount)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(fsopen, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(fsopen)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(fsconfig, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(fsconfig)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(fsmount, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(fsmount)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(fspick, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(fspick)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(io_uring_setup, UnimplementedSyscallSafe);
|
||||
REGISTER_SYSCALL_IMPL(io_uring_enter, UnimplementedSyscallSafe);
|
||||
REGISTER_SYSCALL_IMPL(io_uring_register, UnimplementedSyscallSafe);
|
||||
REGISTER_SYSCALL_IMPL(open_tree, UnimplementedSyscallSafe);
|
||||
REGISTER_SYSCALL_IMPL(move_mount, UnimplementedSyscallSafe);
|
||||
REGISTER_SYSCALL_IMPL(fsopen, UnimplementedSyscallSafe);
|
||||
REGISTER_SYSCALL_IMPL(fsconfig, UnimplementedSyscallSafe);
|
||||
REGISTER_SYSCALL_IMPL(fsmount, UnimplementedSyscallSafe);
|
||||
REGISTER_SYSCALL_IMPL(fspick, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 3, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(pidfd_open, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(pidfd_open)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(pidfd_open, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 8, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(pidfd_getfd, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(pidfd_getfd)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(pidfd_getfd, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 12, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(mount_setattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(mount_setattr)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(mount_setattr, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 14, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(quotactl_fd, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(quotactl_fd)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(quotactl_fd, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 13, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(landlock_create_ruleset, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(landlock_create_ruleset)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(landlock_create_ruleset, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 13, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(landlock_add_rule, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(landlock_add_rule)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(landlock_add_rule, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 13, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(landlock_restrict_self, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(landlock_restrict_self)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(landlock_restrict_self, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 14, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(memfd_secret, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough1<SYSCALL_DEF(memfd_secret)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(memfd_secret, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 15, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(process_mrelease, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(process_mrelease)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL(process_mrelease, UnimplementedSyscallSafe);
|
||||
}
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(io_uring_setup, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(io_uring_setup)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(io_uring_enter, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough6<SYSCALL_DEF(io_uring_enter)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(io_uring_register, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(io_uring_register)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(open_tree, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(open_tree)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(move_mount, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(move_mount)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(fsopen, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(fsopen)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(fsconfig, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(fsconfig)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(fsmount, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(fsmount)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(fspick, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(fspick)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(pidfd_open, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(pidfd_open)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(pidfd_getfd, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(pidfd_getfd)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(mount_setattr, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(mount_setattr)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(quotactl_fd, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(quotactl_fd)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(landlock_create_ruleset, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(landlock_create_ruleset)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(landlock_add_rule, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(landlock_add_rule)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(landlock_restrict_self, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(landlock_restrict_self)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(memfd_secret, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough1<SYSCALL_DEF(memfd_secret)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(process_mrelease, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(process_mrelease)>);
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 16, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(futex_waitv, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(futex_waitv)>);
|
||||
@@ -764,18 +716,10 @@ namespace x64 {
|
||||
SyscallPassthrough6<SYSCALL_DEF(pwritev2)>);
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(io_pgetevents, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough6<SYSCALL_DEF(io_pgetevents)>);
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 1, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(pidfd_send_signal, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(pidfd_send_signal)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL_X64(pidfd_send_signal, UnimplementedSyscallSafe);
|
||||
}
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 10, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(process_madvise, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(process_madvise)>);
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL_X64(process_madvise, UnimplementedSyscallSafe);
|
||||
}
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(pidfd_send_signal, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(pidfd_send_signal)>);
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(process_madvise, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(process_madvise)>);
|
||||
if (Handler->IsHostKernelVersionAtLeast(6, 5, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(cachestat, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(cachestat)>);
|
||||
|
||||
@@ -246,19 +246,28 @@ uint64_t HandleNewClone(FEX::HLE::ThreadStateObject* Thread, FEXCore::Context::C
|
||||
return Thread->StatusCode;
|
||||
}
|
||||
|
||||
static int Clone3Fork(uint32_t flags) {
|
||||
struct clone_args cl_args = {
|
||||
.flags = (flags & (CLONE_FS | CLONE_FILES)),
|
||||
.exit_signal = SIGCHLD,
|
||||
};
|
||||
|
||||
return syscall(SYS_clone3, cl_args, sizeof(cl_args));
|
||||
static int CloneFork(uint32_t flags, uint64_t exit_signal) {
|
||||
return ::syscall(SYSCALL_DEF(clone), (flags & (CLONE_FS | CLONE_FILES)) | exit_signal, nullptr, nullptr, nullptr, nullptr);
|
||||
}
|
||||
|
||||
uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::CpuStateFrame* Frame, uint32_t flags, void* stack,
|
||||
size_t StackSize, pid_t* parent_tid, pid_t* child_tid, void* tls) {
|
||||
// Just before we fork, we lock all syscall mutexes so that both processes will end up with a locked mutex
|
||||
uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::CpuStateFrame* Frame, FEX::HLE::clone3_args* args) {
|
||||
const uint64_t flags = args->args.flags;
|
||||
auto stack = reinterpret_cast<const void*>(args->args.stack);
|
||||
const uint64_t stack_size = args->args.stack_size;
|
||||
auto parent_tid = reinterpret_cast<pid_t*>(args->args.parent_tid);
|
||||
auto child_tid = reinterpret_cast<pid_t*>(args->args.child_tid);
|
||||
auto tls = reinterpret_cast<void*>(args->args.tls);
|
||||
const uint64_t exit_signal = args->args.exit_signal;
|
||||
|
||||
// Sanity check flags here.
|
||||
if (args->Type == TypeOfClone::TYPE_CLONE3) {
|
||||
constexpr uint64_t UnsupportedFlags = CLONE_CLEAR_SIGHAND | CLONE_INTO_CGROUP | CLONE_NEWTIME;
|
||||
if (args->args.flags & UnsupportedFlags) {
|
||||
LogMan::Msg::EFmt("fork: Unsupported flags passed. {:#x}", args->args.flags & UnsupportedFlags);
|
||||
}
|
||||
}
|
||||
|
||||
// Just before we fork, we lock all syscall mutexes so that both processes will end up with a locked mutex
|
||||
uint64_t Mask {~0ULL};
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &Mask, sizeof(Mask));
|
||||
|
||||
@@ -275,7 +284,7 @@ uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::Cp
|
||||
|
||||
// XXX: We don't currently support a real `vfork` as it causes problems.
|
||||
// Currently behaves like a fork (with wait after the fact), which isn't correct. Need to find where the problem is
|
||||
Result = Clone3Fork(flags);
|
||||
Result = CloneFork(flags, exit_signal);
|
||||
|
||||
if (Result == 0) {
|
||||
// Close the read end of the pipe.
|
||||
@@ -286,7 +295,7 @@ uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::Cp
|
||||
close(VForkFDs[1]);
|
||||
}
|
||||
} else {
|
||||
Result = Clone3Fork(flags);
|
||||
Result = CloneFork(flags, exit_signal);
|
||||
}
|
||||
const bool IsChild = Result == 0;
|
||||
|
||||
@@ -309,7 +318,7 @@ uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::Cp
|
||||
// Handle child setup now
|
||||
if (stack != nullptr) {
|
||||
// use specified stack
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = reinterpret_cast<uint64_t>(stack) + StackSize;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = reinterpret_cast<uint64_t>(stack) + stack_size;
|
||||
} else {
|
||||
// In the case of fork and nullptr stack then the child uses the same stack space as the parent
|
||||
// Same virtual address, different addressspace
|
||||
@@ -383,13 +392,43 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
FEX_UNREACHABLE;
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(fork, SyscallFlags::DEFAULT, [](FEXCore::Core::CpuStateFrame* Frame) -> uint64_t {
|
||||
return ForkGuest(Frame->Thread, Frame, 0, 0, 0, 0, 0, 0);
|
||||
});
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(fork, SyscallFlags::DEFAULT, ([](FEXCore::Core::CpuStateFrame* Frame) -> uint64_t {
|
||||
FEX::HLE::clone3_args args {.Type = TypeOfClone::TYPE_CLONE2,
|
||||
.args = {
|
||||
.flags = 0,
|
||||
.pidfd = 0,
|
||||
.child_tid = 0,
|
||||
.parent_tid = 0,
|
||||
.exit_signal = SIGCHLD,
|
||||
.stack = 0,
|
||||
.stack_size = 0,
|
||||
.tls = 0,
|
||||
.set_tid = 0,
|
||||
.set_tid_size = 0,
|
||||
.cgroup = 0,
|
||||
}};
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(vfork, SyscallFlags::DEFAULT, [](FEXCore::Core::CpuStateFrame* Frame) -> uint64_t {
|
||||
return ForkGuest(Frame->Thread, Frame, CLONE_VFORK, 0, 0, 0, 0, 0);
|
||||
});
|
||||
return ForkGuest(Frame->Thread, Frame, &args);
|
||||
}));
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(vfork, SyscallFlags::DEFAULT, ([](FEXCore::Core::CpuStateFrame* Frame) -> uint64_t {
|
||||
FEX::HLE::clone3_args args {.Type = TypeOfClone::TYPE_CLONE2,
|
||||
.args = {
|
||||
.flags = CLONE_VFORK,
|
||||
.pidfd = 0,
|
||||
.child_tid = 0,
|
||||
.parent_tid = 0,
|
||||
.exit_signal = SIGCHLD,
|
||||
.stack = 0,
|
||||
.stack_size = 0,
|
||||
.tls = 0,
|
||||
.set_tid = 0,
|
||||
.set_tid_size = 0,
|
||||
.cgroup = 0,
|
||||
}};
|
||||
|
||||
return ForkGuest(Frame->Thread, Frame, &args);
|
||||
}));
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(getpgrp, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame) -> uint64_t {
|
||||
@@ -419,9 +458,10 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
}
|
||||
|
||||
ThreadObject->StatusCode = status;
|
||||
FEX::HLE::_SyscallHandler->TM.StopThread(ThreadObject);
|
||||
|
||||
return 0;
|
||||
FEX::HLE::_SyscallHandler->TM.DestroyThread(ThreadObject, true);
|
||||
syscall(SYSCALL_DEF(exit), status);
|
||||
// This will never be reached
|
||||
std::terminate();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(prctl, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
|
||||
@@ -20,6 +20,5 @@ struct ThreadStateObject;
|
||||
FEX::HLE::ThreadStateObject* CreateNewThread(FEXCore::Context::Context* CTX, FEXCore::Core::CpuStateFrame* Frame, FEX::HLE::clone3_args* args);
|
||||
uint64_t HandleNewClone(FEX::HLE::ThreadStateObject* Thread, FEXCore::Context::Context* CTX, FEXCore::Core::CpuStateFrame* Frame,
|
||||
FEX::HLE::clone3_args* GuestArgs);
|
||||
uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::CpuStateFrame* Frame, uint32_t flags, void* stack,
|
||||
size_t StackSize, pid_t* parent_tid, pid_t* child_tid, void* tls);
|
||||
uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::CpuStateFrame* Frame, FEX::HLE::clone3_args* args);
|
||||
} // namespace FEX::HLE
|
||||
@@ -56,6 +56,14 @@ void ThreadManager::HandleThreadDeletion(FEX::HLE::ThreadStateObject* Thread, bo
|
||||
}
|
||||
|
||||
if (NeedsTLSUninstall) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Sanity check. This can only be called from the owning thread.
|
||||
{
|
||||
const auto pid = ::getpid();
|
||||
const auto tid = FHU::Syscalls::gettid();
|
||||
LOGMAN_THROW_A_FMT(Thread->ThreadInfo.PID == pid && Thread->ThreadInfo.TID == tid, "Can't delete TLS data from a different thread!");
|
||||
}
|
||||
#endif
|
||||
FEXCore::Allocator::UninstallTLSData(Thread->Thread);
|
||||
}
|
||||
|
||||
@@ -76,6 +84,18 @@ void ThreadManager::NotifyPause() {
|
||||
}
|
||||
|
||||
void ThreadManager::Pause() {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Sanity check. This can't be called from an emulation thread.
|
||||
{
|
||||
const auto pid = ::getpid();
|
||||
const auto tid = FHU::Syscalls::gettid();
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
for (auto& Thread : Threads) {
|
||||
LOGMAN_THROW_A_FMT(!(Thread->ThreadInfo.PID == pid && Thread->ThreadInfo.TID == tid), "Can't put threads to sleep from inside "
|
||||
"emulation thread!");
|
||||
}
|
||||
}
|
||||
#endif
|
||||
NotifyPause();
|
||||
WaitForIdle();
|
||||
}
|
||||
@@ -164,6 +184,15 @@ void ThreadManager::Stop(bool IgnoreCurrentThread) {
|
||||
|
||||
void ThreadManager::SleepThread(FEXCore::Context::Context* CTX, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
auto ThreadObject = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Sanity check. This can only be called from the owning thread.
|
||||
{
|
||||
const auto pid = ::getpid();
|
||||
const auto tid = FHU::Syscalls::gettid();
|
||||
LOGMAN_THROW_A_FMT(ThreadObject->ThreadInfo.PID == pid && ThreadObject->ThreadInfo.TID == tid, "Can't delete TLS data from a different "
|
||||
"thread!");
|
||||
}
|
||||
#endif
|
||||
|
||||
--IdleWaitRefCount;
|
||||
IdleWaitCV.notify_all();
|
||||
|
||||
@@ -84,6 +84,15 @@ struct ThreadStateObject : public FEXCore::Allocator::FEXAllocOperators {
|
||||
std::atomic_bool ThreadSleeping {false};
|
||||
FEXCore::InterruptableConditionVariable ThreadPaused;
|
||||
|
||||
// GDB signal information
|
||||
struct GdbInfoStruct {
|
||||
int Signal {};
|
||||
uint64_t SignalPC {};
|
||||
uint64_t GPRs[32];
|
||||
uint64_t PState {};
|
||||
};
|
||||
std::optional<GdbInfoStruct> GdbInfo;
|
||||
|
||||
int StatusCode {};
|
||||
};
|
||||
|
||||
|
||||
@@ -211,9 +211,11 @@ namespace PThreads {
|
||||
int AttachState {};
|
||||
if (pthread_attr_getdetachstate(&Attr, &AttachState) == 0) {
|
||||
if (AttachState == PTHREAD_CREATE_JOINABLE) {
|
||||
pthread_attr_destroy(&Attr);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
pthread_attr_destroy(&Attr);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -80,33 +80,29 @@ void RegisterEpoll(FEX::HLE::SyscallHandler* Handler) {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 11, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_X32(epoll_pwait2,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int epfd, compat_ptr<FEX::HLE::x32::epoll_event32> events,
|
||||
int maxevent, compat_ptr<timespec32> timeout, const uint64_t* sigmask, size_t sigsetsize) -> uint64_t {
|
||||
fextl::vector<struct epoll_event> Events(std::max(0, maxevent));
|
||||
REGISTER_SYSCALL_IMPL_X32(epoll_pwait2,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int epfd, compat_ptr<FEX::HLE::x32::epoll_event32> events, int maxevent,
|
||||
compat_ptr<timespec32> timeout, const uint64_t* sigmask, size_t sigsetsize) -> uint64_t {
|
||||
fextl::vector<struct epoll_event> Events(std::max(0, maxevent));
|
||||
|
||||
struct timespec tp64 {};
|
||||
struct timespec* timed_ptr {};
|
||||
if (timeout) {
|
||||
tp64 = *timeout;
|
||||
timed_ptr = &tp64;
|
||||
struct timespec tp64 {};
|
||||
struct timespec* timed_ptr {};
|
||||
if (timeout) {
|
||||
tp64 = *timeout;
|
||||
timed_ptr = &tp64;
|
||||
}
|
||||
|
||||
uint64_t Result =
|
||||
::syscall(SYSCALL_DEF(epoll_pwait2), epfd, Events.data(), maxevent, timed_ptr, sigmask, sigsetsize);
|
||||
|
||||
if (Result != -1) {
|
||||
FaultSafeUserMemAccess::VerifyIsWritable(events, sizeof(FEX::HLE::x32::epoll_event32) * Result);
|
||||
for (size_t i = 0; i < Result; ++i) {
|
||||
events[i] = Events[i];
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t Result =
|
||||
::syscall(SYSCALL_DEF(epoll_pwait2), epfd, Events.data(), maxevent, timed_ptr, sigmask, sigsetsize);
|
||||
|
||||
if (Result != -1) {
|
||||
FaultSafeUserMemAccess::VerifyIsWritable(events, sizeof(FEX::HLE::x32::epoll_event32) * Result);
|
||||
for (size_t i = 0; i < Result; ++i) {
|
||||
events[i] = Events[i];
|
||||
}
|
||||
}
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL_X32(epoll_pwait2, UnimplementedSyscallSafe);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
}
|
||||
} // namespace FEX::HLE::x32
|
||||
@@ -212,24 +212,20 @@ void RegisterSignals(FEX::HLE::SyscallHandler* Handler) {
|
||||
return Result;
|
||||
});
|
||||
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 1, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_X32(
|
||||
pidfd_send_signal,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int pidfd, int sig, compat_ptr<FEXCore::x86::siginfo_t> info, unsigned int flags) -> uint64_t {
|
||||
siginfo_t* InfoHost_ptr {};
|
||||
siginfo_t InfoHost {};
|
||||
if (info) {
|
||||
FaultSafeUserMemAccess::VerifyIsReadable(info, sizeof(*info));
|
||||
InfoHost = *info;
|
||||
InfoHost_ptr = &InfoHost;
|
||||
}
|
||||
REGISTER_SYSCALL_IMPL_X32(
|
||||
pidfd_send_signal,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int pidfd, int sig, compat_ptr<FEXCore::x86::siginfo_t> info, unsigned int flags) -> uint64_t {
|
||||
siginfo_t* InfoHost_ptr {};
|
||||
siginfo_t InfoHost {};
|
||||
if (info) {
|
||||
FaultSafeUserMemAccess::VerifyIsReadable(info, sizeof(*info));
|
||||
InfoHost = *info;
|
||||
InfoHost_ptr = &InfoHost;
|
||||
}
|
||||
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(pidfd_send_signal), pidfd, sig, InfoHost_ptr, flags);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL_X32(pidfd_send_signal, UnimplementedSyscallSafe);
|
||||
}
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(pidfd_send_signal), pidfd, sig, InfoHost_ptr, flags);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(
|
||||
rt_sigqueueinfo, [](FEXCore::Core::CpuStateFrame* Frame, pid_t pid, int sig, compat_ptr<FEXCore::x86::siginfo_t> info) -> uint64_t {
|
||||
|
||||
@@ -74,26 +74,21 @@ void RegisterEpoll(FEX::HLE::SyscallHandler* Handler) {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
if (Handler->IsHostKernelVersionAtLeast(5, 11, 0)) {
|
||||
REGISTER_SYSCALL_IMPL_X64(epoll_pwait2,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int epfd, FEX::HLE::epoll_event_x86* events, int maxevent,
|
||||
timespec* timeout, const uint64_t* sigmask, size_t sigsetsize) -> uint64_t {
|
||||
fextl::vector<struct epoll_event> Events(std::max(0, maxevent));
|
||||
REGISTER_SYSCALL_IMPL_X64(epoll_pwait2,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int epfd, FEX::HLE::epoll_event_x86* events, int maxevent,
|
||||
timespec* timeout, const uint64_t* sigmask, size_t sigsetsize) -> uint64_t {
|
||||
fextl::vector<struct epoll_event> Events(std::max(0, maxevent));
|
||||
|
||||
uint64_t Result =
|
||||
::syscall(SYSCALL_DEF(epoll_pwait2), epfd, Events.data(), maxevent, timeout, sigmask, sigsetsize);
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(epoll_pwait2), epfd, Events.data(), maxevent, timeout, sigmask, sigsetsize);
|
||||
|
||||
if (Result != -1) {
|
||||
FaultSafeUserMemAccess::VerifyIsWritable(events, sizeof(FEX::HLE::epoll_event_x86) * Result);
|
||||
for (size_t i = 0; i < Result; ++i) {
|
||||
events[i] = Events[i];
|
||||
}
|
||||
if (Result != -1) {
|
||||
FaultSafeUserMemAccess::VerifyIsWritable(events, sizeof(FEX::HLE::epoll_event_x86) * Result);
|
||||
for (size_t i = 0; i < Result; ++i) {
|
||||
events[i] = Events[i];
|
||||
}
|
||||
}
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
} else {
|
||||
REGISTER_SYSCALL_IMPL_X64(epoll_pwait2, UnimplementedSyscallSafe);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
}
|
||||
} // namespace FEX::HLE::x64
|
||||
@@ -52,8 +52,9 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
.Type = TypeOfClone::TYPE_CLONE2,
|
||||
.args =
|
||||
{
|
||||
.flags = flags, // CSIGNAL is contained in here
|
||||
.pidfd = 0, // For clone, pidfd is duplicated here
|
||||
|
||||
.flags = flags & ~CSIGNAL, // This no longer contains CSIGNAL
|
||||
.pidfd = 0, // For clone, pidfd is duplicated here
|
||||
.child_tid = reinterpret_cast<uint64_t>(child_tid),
|
||||
.parent_tid = reinterpret_cast<uint64_t>(parent_tid),
|
||||
.exit_signal = flags & CSIGNAL,
|
||||
|
||||
@@ -49,7 +49,8 @@ BeginSimulation:
|
||||
bl "#SyncThreadContext"
|
||||
ldr x17, [x18, #0x1788] // TEB->ChpeV2CpuAreaInfo
|
||||
ldr x16, [x17, #0x48] // ChpeV2CpuAreaInfo->EmulatorData[3] - DispatcherLoopTopEnterECFillSRA
|
||||
br x16 // DispatcherLoopTopEnterECFillSRA(CPUArea:x17)
|
||||
mov x10, #0 // Zero ENTRY_FILL_SRA_SINGLE_INST_REG to avoid single step
|
||||
br x16 // DispatcherLoopTopEnterECFillSRA(SingleInst:x10, CPUArea:x17)
|
||||
|
||||
// Called into by FEXCore
|
||||
// Expects the target code address in x9
|
||||
|
||||
@@ -415,7 +415,7 @@ static ARM64_NT_CONTEXT StoreStateToPackedECContext(FEXCore::Core::InternalThrea
|
||||
// See HandleGuestException
|
||||
uint32_t EFlags = CTX->ReconstructCompactedEFLAGS(Thread, false, nullptr, 0);
|
||||
ECContext.Cpsr = 0;
|
||||
ECContext.Cpsr |= (EFlags & (1U << FEXCore::X86State::RFLAG_TF_LOC)) ? (1U << 21) : 0;
|
||||
ECContext.Cpsr |= (EFlags & (1U << FEXCore::X86State::RFLAG_TF_RAW_LOC)) ? (1U << 21) : 0;
|
||||
ECContext.Cpsr |= (EFlags & (1U << FEXCore::X86State::RFLAG_OF_RAW_LOC)) ? (1U << 28) : 0;
|
||||
ECContext.Cpsr |= (EFlags & (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC)) ? (1U << 29) : 0;
|
||||
ECContext.Cpsr |= (EFlags & (1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC)) ? (1U << 30) : 0;
|
||||
@@ -435,7 +435,9 @@ static void RethrowGuestException(const EXCEPTION_RECORD& Rec, ARM64_NT_CONTEXT&
|
||||
auto* Args = reinterpret_cast<KiUserExceptionDispatcherStackLayout*>(FEXCore::AlignDown(GuestSp, 64)) - 1;
|
||||
|
||||
LogMan::Msg::DFmt("Reconstructing context");
|
||||
ReconstructThreadState(Thread, Context);
|
||||
if (!IsDispatcherAddress(Context.Pc)) {
|
||||
ReconstructThreadState(Thread, Context);
|
||||
}
|
||||
Args->Context = StoreStateToPackedECContext(Thread, Context.Fpcr, Context.Fpsr);
|
||||
LogMan::Msg::DFmt("pc: {:X} rip: {:X}", Context.Pc, Args->Context.Pc);
|
||||
|
||||
@@ -443,7 +445,7 @@ static void RethrowGuestException(const EXCEPTION_RECORD& Rec, ARM64_NT_CONTEXT&
|
||||
// Current ARM64EC windows can only restore NZCV+SS when returning from an exception and other flags are left untouched from the handler context.
|
||||
// TODO: Can extend wine to support this by mapping the remaining EFlags into reserved cpsr members.
|
||||
uint32_t EFlags = CTX->ReconstructCompactedEFLAGS(Thread, false, nullptr, 0);
|
||||
EFlags &= (1 << FEXCore::X86State::RFLAG_TF_LOC);
|
||||
EFlags &= ~(1 << FEXCore::X86State::RFLAG_TF_RAW_LOC);
|
||||
CTX->SetFlagsFromCompactedEFLAGS(Thread, EFlags);
|
||||
|
||||
Args->Rec = FEX::Windows::HandleGuestException(Fault, Rec, Args->Context.Pc, Args->Context.X8);
|
||||
@@ -465,6 +467,8 @@ public:
|
||||
}
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) override {
|
||||
ProcessPendingCrossProcessEmulatorWork();
|
||||
|
||||
// Manually raise an exeption with the current JIT state packed into a native context, ntdll handles this and
|
||||
// reenters the JIT (see dlls/ntdll/signal_arm64ec.c in wine).
|
||||
uint64_t FPCR, FPSR;
|
||||
@@ -503,12 +507,13 @@ public:
|
||||
} // namespace Exception
|
||||
|
||||
extern "C" void SyncThreadContext(CONTEXT* Context) {
|
||||
ProcessPendingCrossProcessEmulatorWork();
|
||||
auto* Thread = GetCPUArea().ThreadState();
|
||||
// All other EFlags bits are lost when converting to/from an ARM64EC context, so merge them in from the current JIT state.
|
||||
// This is advisable over dropping their values as thread suspend/resume uses this function, and that can happen at any point in guest code.
|
||||
static constexpr uint32_t ECValidEFlagsMask {(1U << FEXCore::X86State::RFLAG_OF_RAW_LOC) | (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC) | (1U << FEXCore::X86State::RFLAG_SF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_TF_LOC)};
|
||||
(1U << FEXCore::X86State::RFLAG_TF_RAW_LOC)};
|
||||
|
||||
uint32_t StateEFlags = CTX->ReconstructCompactedEFLAGS(Thread, false, nullptr, 0);
|
||||
Context->EFlags = (Context->EFlags & ECValidEFlagsMask) | (StateEFlags & ~ECValidEFlagsMask);
|
||||
@@ -547,6 +552,10 @@ NTSTATUS ProcessInit() {
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
CTX->InitCore();
|
||||
InvalidationTracker.emplace(*CTX, Threads);
|
||||
|
||||
auto MainModule = reinterpret_cast<__TEB*>(NtCurrentTeb())->Peb->ImageBaseAddress;
|
||||
InvalidationTracker->HandleImageMap(reinterpret_cast<uint64_t>(MainModule));
|
||||
|
||||
CPUFeatures.emplace(*CTX);
|
||||
|
||||
X64ReturnInstr = ::VirtualAlloc(nullptr, FEXCore::Utils::FEX_PAGE_SIZE, MEM_COMMIT, PAGE_EXECUTE_READWRITE);
|
||||
@@ -585,7 +594,19 @@ bool ResetToConsistentStateImpl(EXCEPTION_RECORD* Exception, CONTEXT* GuestConte
|
||||
|
||||
std::scoped_lock Lock(ThreadCreationMutex);
|
||||
if (InvalidationTracker->HandleRWXAccessViolation(FaultAddress)) {
|
||||
LogMan::Msg::DFmt("Handled self-modifying code: pc: {:X} fault: {:X}", NativeContext->Pc, FaultAddress);
|
||||
if (CTX->IsAddressInCodeBuffer(CPUArea.ThreadState(), NativeContext->Pc) && !CTX->IsCurrentBlockSingleInst(CPUArea.ThreadState()) &&
|
||||
CTX->IsAddressInCurrentBlock(CPUArea.ThreadState(), FaultAddress, 8)) {
|
||||
// If we are not patching ourself (single inst block case) and patching the current block, this is inline SMC. Reconstruct the current context (before the SMC write) then single step the write to reduce it to regular SMC.
|
||||
Exception::ReconstructThreadState(CPUArea.ThreadState(), *NativeContext);
|
||||
LogMan::Msg::DFmt("Handled inline self-modifying code: pc: {:X} rip: {:X} fault: {:X}", NativeContext->Pc,
|
||||
CPUArea.ThreadState()->CurrentFrame->State.rip, FaultAddress);
|
||||
NativeContext->Pc = CPUArea.DispatcherLoopTopEnterECFillSRA();
|
||||
NativeContext->Sp = CPUArea.EmulatorStackBase();
|
||||
NativeContext->X10 = 1; // Set ENTRY_FILL_SRA_SINGLE_INST_REG to force a single step
|
||||
} else {
|
||||
LogMan::Msg::DFmt("Handled self-modifying code: pc: {:X} fault: {:X}", NativeContext->Pc, FaultAddress);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -694,6 +715,12 @@ void NotifyMemoryProtect(void* Address, SIZE_T Size, ULONG NewProt, BOOL After,
|
||||
}
|
||||
|
||||
NTSTATUS NotifyMapViewOfSection(void* Unk1, void* Address, void* Unk2, SIZE_T Size, ULONG AllocType, ULONG Prot) {
|
||||
if (!InvalidationTracker || !GetCPUArea().ThreadState()) {
|
||||
return STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
std::scoped_lock Lock(ThreadCreationMutex);
|
||||
InvalidationTracker->HandleImageMap(reinterpret_cast<uint64_t>(Address));
|
||||
return STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -789,20 +816,22 @@ NTSTATUS ThreadTerm(HANDLE Thread, LONG ExitCode) {
|
||||
auto* OldThreadState = CPUArea.ThreadState();
|
||||
CPUArea.ThreadState() = nullptr;
|
||||
|
||||
{
|
||||
THREAD_BASIC_INFORMATION Info;
|
||||
if (NTSTATUS Err = NtQueryInformationThread(Thread, ThreadBasicInformation, &Info, sizeof(Info), nullptr); Err) {
|
||||
return Err;
|
||||
}
|
||||
THREAD_BASIC_INFORMATION Info;
|
||||
if (NTSTATUS Err = NtQueryInformationThread(Thread, ThreadBasicInformation, &Info, sizeof(Info), nullptr); Err) {
|
||||
return Err;
|
||||
}
|
||||
|
||||
const auto ThreadTID = reinterpret_cast<uint64_t>(Info.ClientId.UniqueThread);
|
||||
const auto ThreadTID = reinterpret_cast<uint64_t>(Info.ClientId.UniqueThread);
|
||||
{
|
||||
std::scoped_lock Lock(ThreadCreationMutex);
|
||||
Threads.erase(ThreadTID);
|
||||
}
|
||||
|
||||
CTX->DestroyThread(OldThreadState);
|
||||
::VirtualFree(reinterpret_cast<void*>(GetCPUArea().EmulatorStackLimit()), 0, MEM_RELEASE);
|
||||
FEX::Windows::DeinitCRTThread();
|
||||
::VirtualFree(reinterpret_cast<void*>(CPUArea.EmulatorStackLimit()), 0, MEM_RELEASE);
|
||||
if (ThreadTID == GetCurrentThreadId()) {
|
||||
FEX::Windows::DeinitCRTThread();
|
||||
}
|
||||
return STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
|
||||
@@ -18,14 +18,14 @@ void InvalidationTracker::HandleMemoryProtectionNotification(uint64_t Address, u
|
||||
const auto AlignedBase = Address & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
const auto AlignedSize = (Address - AlignedBase + Size + FEXCore::Utils::FEX_PAGE_SIZE - 1) & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
|
||||
if (Prot & (PAGE_EXECUTE | PAGE_EXECUTE_READ | PAGE_EXECUTE_READWRITE)) {
|
||||
if (Prot & (PAGE_EXECUTE | PAGE_EXECUTE_READ | PAGE_EXECUTE_READWRITE | PAGE_EXECUTE_WRITECOPY)) {
|
||||
std::scoped_lock Lock(CTX.GetCodeInvalidationMutex());
|
||||
for (auto Thread : Threads) {
|
||||
CTX.InvalidateGuestCodeRange(Thread.second, AlignedBase, AlignedSize);
|
||||
}
|
||||
}
|
||||
|
||||
if (Prot & PAGE_EXECUTE_READWRITE) {
|
||||
if (Prot & (PAGE_EXECUTE_WRITECOPY | PAGE_EXECUTE_READWRITE)) {
|
||||
LogMan::Msg::DFmt("Add SMC interval: {:X} - {:X}", AlignedBase, AlignedBase + AlignedSize);
|
||||
std::scoped_lock Lock(RWXIntervalsLock);
|
||||
RWXIntervals.Insert({AlignedBase, AlignedBase + AlignedSize});
|
||||
@@ -35,6 +35,21 @@ void InvalidationTracker::HandleMemoryProtectionNotification(uint64_t Address, u
|
||||
}
|
||||
}
|
||||
|
||||
void InvalidationTracker::HandleImageMap(uint64_t Address) {
|
||||
auto* Nt = RtlImageNtHeader(reinterpret_cast<HMODULE>(Address));
|
||||
auto* SectionsBegin = IMAGE_FIRST_SECTION(Nt);
|
||||
auto* SectionsEnd = SectionsBegin + Nt->FileHeader.NumberOfSections;
|
||||
|
||||
for (auto* Section = SectionsBegin; Section != SectionsEnd; Section++) {
|
||||
if ((Section->Characteristics & IMAGE_SCN_MEM_EXECUTE) && (Section->Characteristics & IMAGE_SCN_MEM_WRITE)) {
|
||||
uint64_t SectionBase = Address + Section->VirtualAddress;
|
||||
LogMan::Msg::DFmt("Add image SMC interval: {:X} - {:X}", SectionBase, SectionBase + Section->Misc.VirtualSize);
|
||||
std::scoped_lock Lock(RWXIntervalsLock);
|
||||
RWXIntervals.Insert({SectionBase, SectionBase + Section->Misc.VirtualSize});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void InvalidationTracker::InvalidateContainingSection(uint64_t Address, bool Free) {
|
||||
MEMORY_BASIC_INFORMATION Info;
|
||||
if (NtQueryVirtualMemory(NtCurrentProcess(), reinterpret_cast<void*>(Address), MemoryBasicInformation, &Info, sizeof(Info), nullptr)) {
|
||||
|
||||
@@ -21,6 +21,7 @@ class InvalidationTracker {
|
||||
public:
|
||||
InvalidationTracker(FEXCore::Context::Context& CTX, const std::unordered_map<DWORD, FEXCore::Core::InternalThreadState*>& Threads);
|
||||
void HandleMemoryProtectionNotification(uint64_t Address, uint64_t Size, ULONG Prot);
|
||||
void HandleImageMap(uint64_t Address);
|
||||
void InvalidateContainingSection(uint64_t Address, bool Free);
|
||||
void InvalidateAlignedInterval(uint64_t Address, uint64_t Size, bool Free);
|
||||
void ReprotectRWXIntervals(uint64_t Address, uint64_t Size);
|
||||
|
||||
@@ -131,9 +131,13 @@ uint64_t GetWowTEB(void* TEB) {
|
||||
*reinterpret_cast<LONG*>(reinterpret_cast<uintptr_t>(TEB) + WowTEBOffsetMemberOffset) + reinterpret_cast<uint64_t>(TEB));
|
||||
}
|
||||
|
||||
bool IsAddressInJit(uint64_t Address) {
|
||||
bool IsDispatcherAddress(uint64_t Address) {
|
||||
const auto& Config = SignalDelegator->GetConfig();
|
||||
if (Address >= Config.DispatcherBegin && Address < Config.DispatcherEnd) {
|
||||
return Address >= Config.DispatcherBegin && Address < Config.DispatcherEnd;
|
||||
}
|
||||
|
||||
bool IsAddressInJit(uint64_t Address) {
|
||||
if (IsDispatcherAddress(Address)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -272,7 +276,9 @@ void ReconstructThreadState(CONTEXT* Context) {
|
||||
}
|
||||
|
||||
WOW64_CONTEXT ReconstructWowContext(CONTEXT* Context) {
|
||||
ReconstructThreadState(Context);
|
||||
if (!IsDispatcherAddress(Context->Pc)) {
|
||||
ReconstructThreadState(Context);
|
||||
}
|
||||
|
||||
WOW64_CONTEXT WowContext {
|
||||
.ContextFlags = WOW64_CONTEXT_ALL,
|
||||
@@ -469,6 +475,10 @@ void BTCpuProcessInit() {
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
CTX->InitCore();
|
||||
InvalidationTracker.emplace(*CTX, Threads);
|
||||
|
||||
auto MainModule = reinterpret_cast<__TEB*>(NtCurrentTeb())->Peb->ImageBaseAddress;
|
||||
InvalidationTracker->HandleImageMap(reinterpret_cast<uint64_t>(MainModule));
|
||||
|
||||
CPUFeatures.emplace(*CTX);
|
||||
|
||||
// Allocate the syscall/unixcall trampolines in the lower 2GB of the address space
|
||||
@@ -506,19 +516,21 @@ void BTCpuThreadTerm(HANDLE Thread, LONG ExitCode) {
|
||||
return;
|
||||
}
|
||||
|
||||
{
|
||||
THREAD_BASIC_INFORMATION Info;
|
||||
if (NTSTATUS Err = NtQueryInformationThread(Thread, ThreadBasicInformation, &Info, sizeof(Info), nullptr); Err) {
|
||||
return;
|
||||
}
|
||||
THREAD_BASIC_INFORMATION Info;
|
||||
if (NTSTATUS Err = NtQueryInformationThread(Thread, ThreadBasicInformation, &Info, sizeof(Info), nullptr); Err) {
|
||||
return;
|
||||
}
|
||||
|
||||
const auto ThreadTID = reinterpret_cast<uint64_t>(Info.ClientId.UniqueThread);
|
||||
const auto ThreadTID = reinterpret_cast<uint64_t>(Info.ClientId.UniqueThread);
|
||||
{
|
||||
std::scoped_lock Lock(ThreadCreationMutex);
|
||||
Threads.erase(ThreadTID);
|
||||
}
|
||||
|
||||
CTX->DestroyThread(TLS.ThreadState());
|
||||
FEX::Windows::DeinitCRTThread();
|
||||
if (ThreadTID == GetCurrentThreadId()) {
|
||||
FEX::Windows::DeinitCRTThread();
|
||||
}
|
||||
}
|
||||
|
||||
void* BTCpuGetBopCode() {
|
||||
@@ -710,7 +722,7 @@ bool BTCpuResetToConsistentStateImpl(EXCEPTION_POINTERS* Ptrs) {
|
||||
auto& Fault = Thread->CurrentFrame->SynchronousFaultData;
|
||||
*Exception = FEX::Windows::HandleGuestException(Fault, *Exception, WowContext.Eip, WowContext.Eax);
|
||||
if (Exception->ExceptionCode == EXCEPTION_SINGLE_STEP) {
|
||||
WowContext.EFlags &= ~(1 << FEXCore::X86State::RFLAG_TF_LOC);
|
||||
WowContext.EFlags &= ~(1 << FEXCore::X86State::RFLAG_TF_RAW_LOC);
|
||||
}
|
||||
// wow64.dll will handle adjusting PC in the dispatched context after a breakpoint
|
||||
|
||||
|
||||
@@ -477,6 +477,7 @@ NTSTATUS WINAPI RtlDeleteCriticalSection(RTL_CRITICAL_SECTION*);
|
||||
NTSTATUS WINAPI RtlEnterCriticalSection(RTL_CRITICAL_SECTION*);
|
||||
ULONG WINAPI RtlFindClearBitsAndSet(PRTL_BITMAP, ULONG, ULONG);
|
||||
ULONG WINAPI RtlGetCurrentDirectory_U(ULONG, LPWSTR);
|
||||
PIMAGE_NT_HEADERS WINAPI RtlImageNtHeader(HMODULE);
|
||||
PVOID WINAPI RtlImageDirectoryEntryToData(HMODULE, BOOL, WORD, ULONG*);
|
||||
void WINAPI RtlInitializeConditionVariable(RTL_CONDITION_VARIABLE*);
|
||||
NTSTATUS WINAPI RtlInitializeCriticalSection(RTL_CRITICAL_SECTION*);
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
#include "diagnostics.h"
|
||||
|
||||
#include <clang/AST/RecursiveASTVisitor.h>
|
||||
#include <clang/Basic/Version.h>
|
||||
#include <clang/Frontend/CompilerInstance.h>
|
||||
|
||||
#include <fmt/format.h>
|
||||
@@ -13,6 +14,14 @@ struct NamespaceAnnotations {
|
||||
bool indirect_guest_calls = false;
|
||||
};
|
||||
|
||||
static clang::SourceLocation GetTemplateArgLocation(clang::ClassTemplateSpecializationDecl* decl, unsigned i) {
|
||||
#if CLANG_VERSION_MAJOR >= 19
|
||||
return decl->getTemplateArgsAsWritten()->getTemplateArgs()[i].getLocation();
|
||||
#else
|
||||
return decl->getTypeAsWritten()->getTypeLoc().getAs<clang::TemplateSpecializationTypeLoc>().getArgLoc(i).getLocation();
|
||||
#endif
|
||||
}
|
||||
|
||||
static NamespaceAnnotations GetNamespaceAnnotations(clang::ASTContext& context, clang::CXXRecordDecl* decl) {
|
||||
if (!decl->hasDefinition()) {
|
||||
return {};
|
||||
@@ -245,8 +254,7 @@ void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
type = type->getLocallyUnqualifiedSingleStepDesugaredType();
|
||||
|
||||
if (param_idx >= function->getNumParams() || param_idx < -1) {
|
||||
throw report_error(decl->getTypeAsWritten()->getTypeLoc().getAs<clang::TemplateSpecializationTypeLoc>().getArgLoc(1).getLocation(),
|
||||
"Out-of-bounds parameter index passed to fex_gen_param");
|
||||
throw report_error(GetTemplateArgLocation(decl, 1), "Out-of-bounds parameter index passed to fex_gen_param");
|
||||
}
|
||||
|
||||
auto expected_type = param_idx == -1 ? function->getReturnType() : function->getParamDecl(param_idx)->getType();
|
||||
@@ -254,8 +262,7 @@ void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
if (!type->isVoidType() && !context.hasSameType(type, expected_type)) {
|
||||
auto loc = param_idx == -1 ? function->getReturnTypeSourceRange().getBegin() :
|
||||
function->getParamDecl(param_idx)->getTypeSourceInfo()->getTypeLoc().getBeginLoc();
|
||||
throw report_error(decl->getTypeAsWritten()->getTypeLoc().getAs<clang::TemplateSpecializationTypeLoc>().getArgLoc(2).getLocation(),
|
||||
"Type passed to fex_gen_param doesn't match the function signature")
|
||||
throw report_error(GetTemplateArgLocation(decl, 2), "Type passed to fex_gen_param doesn't match the function signature")
|
||||
.addNote(report_error(loc, "Expected this type instead"));
|
||||
}
|
||||
|
||||
@@ -296,8 +303,7 @@ void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
const auto& template_args = decl->getTemplateArgs();
|
||||
assert(template_args.size() == 1);
|
||||
|
||||
const auto template_arg_loc =
|
||||
decl->getTypeAsWritten()->getTypeLoc().castAs<clang::TemplateSpecializationTypeLoc>().getArgLoc(0).getLocation();
|
||||
const auto template_arg_loc = GetTemplateArgLocation(decl, 0);
|
||||
|
||||
if (auto emitted_function = llvm::dyn_cast<clang::FunctionDecl>(template_args[0].getAsDecl())) {
|
||||
// Process later
|
||||
@@ -333,8 +339,7 @@ void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
const auto& template_args = decl->getTemplateArgs();
|
||||
assert(template_args.size() == 1);
|
||||
|
||||
const auto template_arg_loc =
|
||||
decl->getTypeAsWritten()->getTypeLoc().castAs<clang::TemplateSpecializationTypeLoc>().getArgLoc(0).getLocation();
|
||||
const auto template_arg_loc = GetTemplateArgLocation(decl, 0);
|
||||
|
||||
if (auto emitted_function = llvm::dyn_cast<clang::FunctionDecl>(template_args[0].getAsDecl())) {
|
||||
auto return_type = emitted_function->getReturnType();
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include "llvm/Support/Signals.h"
|
||||
|
||||
#include <iostream>
|
||||
#include <optional>
|
||||
#include <string>
|
||||
|
||||
#include "interface.h"
|
||||
@@ -17,7 +18,7 @@ void print_usage(const char* program_name) {
|
||||
int main(int argc, char* const argv[]) {
|
||||
llvm::sys::PrintStackTraceOnErrorSignal(argv[0]);
|
||||
|
||||
if (argc < 5) {
|
||||
if (argc < 6) {
|
||||
print_usage(argv[0]);
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
@@ -32,7 +33,7 @@ int main(int argc, char* const argv[]) {
|
||||
}
|
||||
|
||||
// Process arguments before the "--" separator
|
||||
if (argc != 5 && argc != 6) {
|
||||
if (argc != 6 && argc != 7) {
|
||||
print_usage(argv[0]);
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
@@ -42,6 +43,7 @@ int main(int argc, char* const argv[]) {
|
||||
const std::string libname = *arg++;
|
||||
const std::string target_abi = *arg++;
|
||||
const std::string output_filename = *arg++;
|
||||
const std::string x86_rootfs = *arg++;
|
||||
|
||||
OutputFilenames output_filenames;
|
||||
if (target_abi == "-host") {
|
||||
@@ -65,27 +67,52 @@ int main(int argc, char* const argv[]) {
|
||||
|
||||
ClangTool GuestTool = Tool;
|
||||
|
||||
{
|
||||
const bool is_32bit_guest = (argv[5] == std::string_view {"-for-32bit-guest"});
|
||||
auto append_guest_args = [is_32bit_guest](const clang::tooling::CommandLineArguments& Args, clang::StringRef) {
|
||||
clang::tooling::CommandLineArguments AdjustedArgs = Args;
|
||||
const char* platform = is_32bit_guest ? "i686" : "x86_64";
|
||||
if (is_32bit_guest) {
|
||||
AdjustedArgs.push_back("-m32");
|
||||
AdjustedArgs.push_back("-DIS_32BIT_THUNK");
|
||||
}
|
||||
AdjustedArgs.push_back(std::string {"--target="} + platform + "-linux-unknown");
|
||||
AdjustedArgs.push_back("-isystem");
|
||||
AdjustedArgs.push_back(std::string {"/usr/"} + platform + "-linux-gnu/include/");
|
||||
AdjustedArgs.push_back("-DGUEST_THUNK_LIBRARY");
|
||||
return AdjustedArgs;
|
||||
};
|
||||
GuestTool.appendArgumentsAdjuster(append_guest_args);
|
||||
}
|
||||
auto append_x86_rootfs_includes = [&x86_rootfs](clang::tooling::CommandLineArguments& Args, const char* triple) {
|
||||
if (x86_rootfs == "/") {
|
||||
return;
|
||||
}
|
||||
|
||||
Args.push_back("--sysroot");
|
||||
Args.push_back(x86_rootfs);
|
||||
|
||||
// The dev rootfs is only really needed for the standard library.
|
||||
// Other libraries generally don't have platform specific headers.
|
||||
Args.push_back("-idirafter");
|
||||
Args.push_back("/usr/include/");
|
||||
};
|
||||
|
||||
// Analyse data layout for guest ABI
|
||||
const bool is_32bit_guest = (argv[6] == std::string_view {"-for-32bit-guest"});
|
||||
GuestTool.appendArgumentsAdjuster([&](const clang::tooling::CommandLineArguments& Args, clang::StringRef) {
|
||||
clang::tooling::CommandLineArguments AdjustedArgs = Args;
|
||||
const char* platform = is_32bit_guest ? "i686-linux-gnu" : "x86_64-linux-gnu";
|
||||
if (is_32bit_guest) {
|
||||
AdjustedArgs.push_back("-m32");
|
||||
AdjustedArgs.push_back("-DIS_32BIT_THUNK");
|
||||
}
|
||||
AdjustedArgs.push_back("-DGUEST_THUNK_LIBRARY");
|
||||
AdjustedArgs.push_back(std::string {"--target="} + platform);
|
||||
AdjustedArgs.push_back("-isystem");
|
||||
AdjustedArgs.push_back(std::string {"/usr/"} + platform + "/include/");
|
||||
|
||||
append_x86_rootfs_includes(AdjustedArgs, platform);
|
||||
|
||||
return AdjustedArgs;
|
||||
});
|
||||
auto data_layout_analysis_factory = std::make_unique<AnalyzeDataLayoutActionFactory>();
|
||||
GuestTool.run(data_layout_analysis_factory.get());
|
||||
auto& data_layout = data_layout_analysis_factory->GetDataLayout();
|
||||
|
||||
// Run generator for target ABI
|
||||
Tool.appendArgumentsAdjuster([&](const clang::tooling::CommandLineArguments& Args, clang::StringRef) {
|
||||
clang::tooling::CommandLineArguments AdjustedArgs = Args;
|
||||
AdjustedArgs.push_back("-DIS_HOST_THUNKGEN_PASS");
|
||||
if (target_abi == "-guest") {
|
||||
const char* platform = is_32bit_guest ? "i686-linux-gnu" : "x86_64-linux-gnu";
|
||||
append_x86_rootfs_includes(AdjustedArgs, platform);
|
||||
}
|
||||
|
||||
return AdjustedArgs;
|
||||
});
|
||||
return Tool.run(std::make_unique<GenerateThunkLibsActionFactory>(std::move(libname), std::move(output_filenames), data_layout).get());
|
||||
}
|
||||
@@ -9,6 +9,10 @@ if (ENABLE_CLANG_THUNKS)
|
||||
add_link_options(${LD_OVERRIDE})
|
||||
endif()
|
||||
|
||||
if (NOT X86_DEV_ROOTFS)
|
||||
message(FATAL_ERROR "X86_DEV_ROOTFS must be set (use \"/\" to ignore)")
|
||||
endif()
|
||||
|
||||
find_program(CCACHE_PROGRAM ccache)
|
||||
if(CCACHE_PROGRAM)
|
||||
message(STATUS "CCache enabled for guest thunks")
|
||||
@@ -77,7 +81,7 @@ function(generate NAME SOURCE_FILE)
|
||||
OUTPUT "${OUTFILE}"
|
||||
DEPENDS "${GENERATOR_EXE}"
|
||||
DEPENDS "${SOURCE_FILE}"
|
||||
COMMAND "${GENERATOR_EXE}" "${SOURCE_FILE}" "${NAME}" "-guest" "${OUTFILE}" ${BITNESS_FLAGS} -- -std=c++20 ${BITNESS_FLAGS2}
|
||||
COMMAND "${GENERATOR_EXE}" "${SOURCE_FILE}" "${NAME}" "-guest" "${OUTFILE}" "${X86_DEV_ROOTFS}" ${BITNESS_FLAGS} -- -std=c++20 ${BITNESS_FLAGS2}
|
||||
# Expand compile definitions to space-separated list of -D parameters
|
||||
"$<$<BOOL:${compile_prop}>:;-D$<JOIN:${compile_prop},;-D>>"
|
||||
# Expand include directories to space-separated list of -isystem parameters
|
||||
|
||||
@@ -53,7 +53,7 @@ function(generate NAME SOURCE_FILE GUEST_BITNESS)
|
||||
OUTPUT "${OUTFILE}"
|
||||
DEPENDS "${SOURCE_FILE}"
|
||||
DEPENDS thunkgen
|
||||
COMMAND thunkgen "${SOURCE_FILE}" "${NAME}" "-host" "${OUTFILE}" ${BITNESS_FLAGS} -- -std=c++20
|
||||
COMMAND thunkgen "${SOURCE_FILE}" "${NAME}" "-host" "${OUTFILE}" "${X86_DEV_ROOTFS}" ${BITNESS_FLAGS} -- -std=c++20
|
||||
# Expand compile definitions to space-separated list of -D parameters
|
||||
"$<$<BOOL:${compile_prop}>:;-D$<JOIN:${compile_prop},;-D>>"
|
||||
# Expand include directories to space-separated list of -isystem parameters
|
||||
|
||||
@@ -40,12 +40,6 @@ Follow the steps in: https://github.com/FEX-Emu/FEX-ppa/blob/main/README.md
|
||||
* Requires PPA GPG key signing access
|
||||
* Wait the 20-30 minutes for Ubuntu PPA to build and publish the binaries
|
||||
|
||||
## ArchLinux AUR package
|
||||
* Clone https://aur.archlinux.org/packages/fex-emu
|
||||
* Requires maintainer or co-maintainer permissions
|
||||
* Update PKGBUILD and .SRCINFO file.
|
||||
* Push changes upstream
|
||||
|
||||
## Github releases page Steps
|
||||
* Requires administrative rights
|
||||
* Go to https://github.com/FEX-Emu/FEX/releases
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# FEX-2412
|
||||
# FEX-2501
|
||||
|
||||
## FEXCore
|
||||
See [FEXCore/Readme.md](../FEXCore/Readme.md) for more details
|
||||
|
||||
@@ -1,3 +1,15 @@
|
||||
# Simulator on x86 doesn't pass these tests due to not using float128
|
||||
Test_X87/precision_test_fcos.asm
|
||||
Test_X87/precision_test_fsin.asm
|
||||
Test_X87/precision_test_ftan.asm
|
||||
Test_X87/precision_test_fatan.asm
|
||||
Test_X87/precision_test_fyl2xp1.asm
|
||||
Test_X87/precision_test_neg_fcos.asm
|
||||
Test_X87/precision_test_neg_fsin.asm
|
||||
Test_X87/precision_test_neg_ftan.asm
|
||||
Test_X87/precision_test_neg_fatan.asm
|
||||
Test_X87/precision_test_neg_fyl2xp1.asm
|
||||
|
||||
# AES unsupported in simulator
|
||||
Test_H0F38/66_DB.asm
|
||||
Test_H0F38/66_DC.asm
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"MM0": "0x3f800000bf800000"
|
||||
},
|
||||
"HostFeatures": ["3DNOW"]
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX-Emu had a bug with 3DNow! ModRM decoding when the source was SIB encoded.
|
||||
; This would result in a crash in the frontend instruction decoding.
|
||||
; Generate a 3DNow! instruction that uses SIB encoding to ensure this code path is tested.
|
||||
lea rax, [rel data1]
|
||||
mov rbx, 0
|
||||
pi2fw mm0, [rbx * 8 + rax + 0]
|
||||
|
||||
hlt
|
||||
|
||||
align 8
|
||||
data1:
|
||||
dw -1
|
||||
dw 0xFF
|
||||
dw 1
|
||||
dw 0xFF
|
||||
@@ -0,0 +1,34 @@
|
||||
%ifdef CONFIG
|
||||
{}
|
||||
%endif
|
||||
|
||||
; FEX-Emu had a bug in decoding the H0F3A instruction table.
|
||||
; It would accidentally require REX.W to not be set on the suite of instructions that ignore the flag.
|
||||
; This just executes all instructions from H0F3A that ignore the REX.W flag, to ensure it decodes.
|
||||
|
||||
o64 palignr mm0, mm1, 0
|
||||
o64 roundps xmm0, xmm1, 0
|
||||
o64 roundpd xmm0, xmm1, 0
|
||||
o64 roundss xmm0, xmm1, 0
|
||||
o64 roundsd xmm0, xmm1, 0
|
||||
o64 blendps xmm0, xmm1, 0
|
||||
o64 blendpd xmm0, xmm1, 0
|
||||
o64 palignr xmm0, xmm1, 0
|
||||
o64 pextrb eax, xmm0, 0
|
||||
o64 pextrw eax, xmm0, 0
|
||||
o64 extractps eax, xmm0, 0
|
||||
o64 extractps eax, xmm0, 0
|
||||
o64 pinsrb xmm0, eax, 0
|
||||
o64 insertps xmm0, xmm1, 0
|
||||
o64 dpps xmm0, xmm1, 0
|
||||
o64 dppd xmm0, xmm1, 0
|
||||
o64 mpsadbw xmm0, xmm1, 0
|
||||
o64 pclmulqdq xmm0, xmm1, 0
|
||||
o64 pcmpestrm xmm0, xmm1, 0
|
||||
o64 pcmpestri xmm0, xmm1, 0
|
||||
o64 pcmpistrm xmm0, xmm1, 0
|
||||
o64 pcmpistri xmm0, xmm1, 0
|
||||
o64 sha1rnds4 xmm0, xmm1, 0
|
||||
o64 aeskeygenassist xmm0, xmm1, 0
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,26 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x000000000f0f0f0f",
|
||||
"RBX": "0x000000000f0f0f0f",
|
||||
"RCX": "0x000000000f0f0f0f",
|
||||
"RDX": "0x00000000ffffffff",
|
||||
"R9": "0x000000000f0f0f0f"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX had several bugs in its constprop pass where 32->64 bit truncation behaviour wasn't accounted for leading
|
||||
; to incorrectly inserting instead.
|
||||
|
||||
mov rax, 0x0f0f0f0f0f0f0f0f
|
||||
mov rbx, 0x0f0f0f0f0f0f0f0f
|
||||
mov rcx, 0x0f0f0f0f0f0f0f0f
|
||||
mov rdx, 1
|
||||
mov r9, 0x0f0f0f0f0f0f0f0f
|
||||
xor eax, 0
|
||||
and ebx, ebx
|
||||
shr ecx, 0
|
||||
neg edx
|
||||
shl r9d, 0
|
||||
hlt
|
||||
@@ -0,0 +1,34 @@
|
||||
%ifndef X87_CW_INC
|
||||
%define X87_CW_INC
|
||||
|
||||
; Sets x87 precision and rounding modes
|
||||
; Uses the stack and clobbers rax
|
||||
; Args: precision constant, rounding constant
|
||||
%macro set_cw_precision_rounding 2
|
||||
sub rsp, 2
|
||||
fnstcw [rsp]
|
||||
movzx ax, [rsp]
|
||||
|
||||
; Precision
|
||||
and eax, ~(3 << 8)
|
||||
or eax, %1 << 8
|
||||
|
||||
; Rounding
|
||||
and eax, ~(3 << 10)
|
||||
or eax, %2 << 10
|
||||
|
||||
mov [rsp], ax
|
||||
fldcw [rsp]
|
||||
add rsp, 2
|
||||
%endmacro
|
||||
|
||||
x87_prec_32 equ 00b
|
||||
x87_prec_64 equ 10b
|
||||
x87_prec_80 equ 11b
|
||||
|
||||
x87_round_nearest equ 00b
|
||||
x87_round_down equ 01b
|
||||
x87_round_up equ 10b
|
||||
x87_round_towards_zero equ 11b
|
||||
|
||||
%endif
|
||||
@@ -2,7 +2,9 @@
|
||||
{
|
||||
"RegData": {
|
||||
"MM0": "0x0000000200000001",
|
||||
"MM1": "0xFFFFFFFEFFFFFFFF"
|
||||
"MM1": "0xFFFFFFFEFFFFFFFF",
|
||||
"MM2": "0x8000000080000000",
|
||||
"MM3": "0x8000000080000000"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -29,6 +31,16 @@ mov [rdx + 8 * 4], rax
|
||||
mov rax, 0x5152535455565758
|
||||
mov [rdx + 8 * 5], rax
|
||||
|
||||
mov rax, 0x7ff0000000000000
|
||||
mov [rdx + 8 * 6], rax
|
||||
mov rax, 0xfff0000000000000
|
||||
mov [rdx + 8 * 7], rax
|
||||
|
||||
mov rax, 0x7ff8000000000000
|
||||
mov [rdx + 8 * 8], rax
|
||||
mov rax, 0x7fefffffffffffff
|
||||
mov [rdx + 8 * 9], rax
|
||||
|
||||
movq mm0, [rdx + 8 * 4]
|
||||
movq mm1, [rdx + 8 * 4]
|
||||
|
||||
@@ -36,5 +48,7 @@ movapd xmm2, [rdx + 8 * 0]
|
||||
|
||||
cvtpd2pi mm0, xmm2
|
||||
cvtpd2pi mm1, [rdx + 8 * 2]
|
||||
cvtpd2pi mm2, [rdx + 8 * 6]
|
||||
cvtpd2pi mm3, [rdx + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -2,7 +2,9 @@
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0x404000003F800000", "0x0"],
|
||||
"XMM1": ["0x3FF0000000000000", "0x4008000000000000"]
|
||||
"XMM1": ["0x3FF0000000000000", "0x4008000000000000"],
|
||||
"XMM2": ["0xff8000007f800000", "0x0000000000000000"],
|
||||
"XMM3": ["0x7f8000007fc00000", "0x0000000000000000"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -19,9 +21,22 @@ mov [rdx + 8 * 2], rax
|
||||
mov rax, 0xFFFFFFFFFFFFFFFF
|
||||
mov [rdx + 8 * 3], rax
|
||||
|
||||
mov rax, 0x7ff0000000000000
|
||||
mov [rdx + 8 * 4], rax
|
||||
mov rax, 0xfff0000000000000
|
||||
mov [rdx + 8 * 5], rax
|
||||
mov rax, 0x7ff8000000000000
|
||||
mov [rdx + 8 * 6], rax
|
||||
mov rax, 0x7fefffffffffffff
|
||||
mov [rdx + 8 * 7], rax
|
||||
|
||||
movapd xmm0, [rdx + 8 * 2]
|
||||
movapd xmm1, [rdx]
|
||||
movapd xmm2, [rdx + 8 * 4]
|
||||
movapd xmm3, [rdx + 8 * 6]
|
||||
|
||||
cvtpd2ps xmm0, xmm1
|
||||
cvtpd2ps xmm2, xmm2
|
||||
cvtpd2ps xmm3, xmm3
|
||||
|
||||
hlt
|
||||
@@ -2,7 +2,8 @@
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0x0000000100000001", "0x0000000200000002"],
|
||||
"XMM1": ["0x0000000400000004", "0x0000000800000008"]
|
||||
"XMM1": ["0x0000000400000004", "0x0000000800000008"],
|
||||
"XMM2": ["0x8000000000000000", "0x8000000080000000"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -24,6 +25,11 @@ mov [rdx + 8 * 4], rax
|
||||
mov rax, 0x5152535455565758
|
||||
mov [rdx + 8 * 5], rax
|
||||
|
||||
mov rax, 0x7fc000007f800000
|
||||
mov [rdx + 8 * 6], rax
|
||||
mov rax, 0xff800000ff7fffee
|
||||
mov [rdx + 8 * 7], rax
|
||||
|
||||
; Set up MXCSR to truncate
|
||||
mov eax, 0x7F80
|
||||
mov [rdx + 8 * 6], eax
|
||||
@@ -36,5 +42,6 @@ movapd xmm2, [rdx + 8 * 0]
|
||||
|
||||
cvtps2dq xmm0, xmm2
|
||||
cvtps2dq xmm1, [rdx + 8 * 2]
|
||||
cvtps2dq xmm2, [rdx + 8 * 6]
|
||||
|
||||
hlt
|
||||
@@ -23,7 +23,8 @@ mov [rdx + 8 * 5], rax
|
||||
|
||||
movapd xmm2, [rdx + 8 * 0]
|
||||
|
||||
movq xmm0, xmm2
|
||||
; movq xmm0, xmm2
|
||||
db 0x66, 0x0f, 0xd6, 11_010_000b
|
||||
movq [rdx + 8 * 2], xmm2
|
||||
movapd xmm1, [rdx + 8 * 2]
|
||||
|
||||
|
||||
@@ -4,7 +4,11 @@
|
||||
"RAX": "0x1",
|
||||
"RBX": "0x2",
|
||||
"RCX": "0x3",
|
||||
"RDX": "0x4"
|
||||
"RDX": "0x4",
|
||||
"R9": "0x8000000000000000",
|
||||
"R10": "0x8000000000000000",
|
||||
"R11": "0x8000000000000000",
|
||||
"R12": "0x8000000000000000"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -31,6 +35,16 @@ mov [rdx + 8 * 6], rax
|
||||
mov rax, 0x5152535455565758
|
||||
mov [rdx + 8 * 7], rax
|
||||
|
||||
mov rax, 0x7ff0000000000000
|
||||
mov [rdx + 8 * 8], rax
|
||||
mov rax, 0xfff0000000000000
|
||||
mov [rdx + 8 * 9], rax
|
||||
mov rax, 0x7ff8000000000000
|
||||
mov [rdx + 8 * 10], rax
|
||||
mov rax, 0x7fefffffffffffff
|
||||
mov [rdx + 8 * 11], rax
|
||||
|
||||
|
||||
movapd xmm0, [rdx + 8 * 0]
|
||||
movapd xmm1, [rdx + 8 * 2]
|
||||
|
||||
@@ -38,6 +52,10 @@ cvtsd2si eax, xmm0
|
||||
cvtsd2si rbx, xmm1
|
||||
|
||||
cvtsd2si ecx, [rdx + 8 * 4]
|
||||
cvtsd2si r9, [rdx + 8 * 8]
|
||||
cvtsd2si r10, [rdx + 8 * 9]
|
||||
cvtsd2si r11, [rdx + 8 * 10]
|
||||
cvtsd2si r12, [rdx + 8 * 11]
|
||||
cvtsd2si rdx, [rdx + 8 * 6]
|
||||
|
||||
hlt
|
||||
@@ -2,7 +2,9 @@
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0x0000000200000001", "0x0"],
|
||||
"XMM1": ["0xFFFFFFFEFFFFFFFF", "0x0"]
|
||||
"XMM1": ["0xFFFFFFFEFFFFFFFF", "0x0"],
|
||||
"XMM2": ["0x8000000080000000", "0x0"],
|
||||
"XMM3": ["0x8000000080000000", "0x0"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -24,6 +26,16 @@ mov [rdx + 8 * 4], rax
|
||||
mov rax, 0x5152535455565758
|
||||
mov [rdx + 8 * 5], rax
|
||||
|
||||
mov rax, 0x7ff0000000000000
|
||||
mov [rdx + 8 * 6], rax
|
||||
mov rax, 0xfff0000000000000
|
||||
mov [rdx + 8 * 7], rax
|
||||
|
||||
mov rax, 0x7ff8000000000000
|
||||
mov [rdx + 8 * 8], rax
|
||||
mov rax, 0x7fefffffffffffff
|
||||
mov [rdx + 8 * 9], rax
|
||||
|
||||
movapd xmm0, [rdx + 8 * 4]
|
||||
movapd xmm1, [rdx + 8 * 4]
|
||||
|
||||
@@ -31,5 +43,7 @@ movapd xmm2, [rdx + 8 * 0]
|
||||
|
||||
cvtpd2dq xmm0, xmm2
|
||||
cvtpd2dq xmm1, [rdx + 8 * 2]
|
||||
cvtpd2dq xmm2, [rdx + 8 * 6]
|
||||
cvtpd2dq xmm3, [rdx + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,29 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"MM0": "0xa2a4a6a8aaacaeb0",
|
||||
"MM1": "0xa2a4a6a8aaacaeb0"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov rdx, 0xe0000000
|
||||
|
||||
mov rax, 0x6162636465666768
|
||||
mov [rdx + 8 * 0], rax
|
||||
mov rax, 0x7172737475767778
|
||||
mov [rdx + 8 * 1], rax
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [rdx + 8 * 2], rax
|
||||
mov rax, 0x5152535455565758
|
||||
mov [rdx + 8 * 3], rax
|
||||
|
||||
movq mm0, [rdx]
|
||||
paddq mm0, [rdx + 8 * 2]
|
||||
|
||||
movq mm1, [rdx]
|
||||
movq mm2, [rdx + 8 * 2]
|
||||
paddq mm1, mm2
|
||||
|
||||
hlt
|
||||
@@ -5,7 +5,8 @@
|
||||
"XMM0": ["0x0000000200000001", "0x0000000000000000", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM1": ["0xFFFFFFFEFFFFFFFF", "0x0000000000000000", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM3": ["0x0000000200000001", "0x0000000200000001", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM4": ["0xFFFFFFFEFFFFFFFF", "0xFFFFFFFEFFFFFFFF", "0x0000000000000000", "0x0000000000000000"]
|
||||
"XMM4": ["0xFFFFFFFEFFFFFFFF", "0xFFFFFFFEFFFFFFFF", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM5": ["0x8000000080000000", "0x8000000080000000", "0x0000000000000000", "0x0000000000000000"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -22,6 +23,8 @@ vcvtpd2dq xmm1, oword [rdx + 32 * 1]
|
||||
vcvtpd2dq xmm3, ymm2
|
||||
vcvtpd2dq xmm4, yword [rdx + 32 * 1]
|
||||
|
||||
vcvtpd2dq xmm5, yword [rdx + 32 * 3]
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
@@ -40,3 +43,8 @@ dq 0x4142434445464748
|
||||
dq 0x5152535455565758
|
||||
dq 0x4142434445464748
|
||||
dq 0x5152535455565758
|
||||
|
||||
dq 0x7ff0000000000000
|
||||
dq 0xfff0000000000000
|
||||
dq 0x7ff8000000000000
|
||||
dq 0x7fefffffffffffff
|
||||
@@ -11,7 +11,8 @@
|
||||
"XMM10": ["0x4214ADB642B062C4", "0x41245B0E42461AA5", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM11": ["0x41A1B712429B697F", "0x42252CF2411CE3BD", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM12": ["0x42662BE34176837B", "0x425E2C0D4119C75A", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM13": ["0x409C30014253A13B", "0x4041495242910EC1", "0x0000000000000000", "0x0000000000000000"]
|
||||
"XMM13": ["0x409C30014253A13B", "0x4041495242910EC1", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM14": ["0xff8000007f800000", "0x7f8000007fc00000", "0x0000000000000000", "0x0000000000000000"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -29,6 +30,7 @@ vcvtpd2ps xmm10, yword [rdx + 32 * 10]
|
||||
vcvtpd2ps xmm11, yword [rdx + 32 * 11]
|
||||
vcvtpd2ps xmm12, yword [rdx + 32 * 12]
|
||||
vcvtpd2ps xmm13, yword [rdx + 32 * 13]
|
||||
vcvtpd2ps xmm14, yword [rdx + 32 * 14]
|
||||
|
||||
hlt
|
||||
|
||||
@@ -75,3 +77,6 @@ dq 9.61117, 55.54302
|
||||
|
||||
dq 52.90745, 4.88086
|
||||
dq 72.52882, 3.0201
|
||||
|
||||
dq 0x7ff0000000000000, 0xfff0000000000000
|
||||
dq 0x7ff8000000000000, 0x7fefffffffffffff
|
||||
@@ -5,7 +5,8 @@
|
||||
"XMM0": ["0x0000000100000001", "0x0000000200000002", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM1": ["0x0000000400000004", "0x0000000800000008", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM3": ["0x0000000100000001", "0x0000000200000002", "0x0000000100000001", "0x0000000200000002"],
|
||||
"XMM4": ["0x0000000400000004", "0x0000000800000008", "0x0000000400000004", "0x0000000800000008"]
|
||||
"XMM4": ["0x0000000400000004", "0x0000000800000008", "0x0000000400000004", "0x0000000800000008"],
|
||||
"XMM5": ["0x8000000080000000", "0x8000000080000000", "0x8000000080000000", "0x0000000000000000"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -24,6 +25,7 @@ vcvtps2dq xmm1, [rdx + 32 * 1]
|
||||
|
||||
vcvtps2dq ymm3, ymm2
|
||||
vcvtps2dq ymm4, [rdx + 32 * 1]
|
||||
vcvtps2dq ymm5, [rdx + 32 * 3]
|
||||
|
||||
hlt
|
||||
|
||||
@@ -44,5 +46,10 @@ dq 0x5152535455565758
|
||||
dq 0x4142434445464748
|
||||
dq 0x5152535455565758
|
||||
|
||||
dq 0x7fc000007f800000
|
||||
dq 0xff800000ff7fffee
|
||||
dq 0x7bc097cefbc097ce
|
||||
dq 0x0000000080000000
|
||||
|
||||
.mxcsr:
|
||||
dq 0x0000000000007F80
|
||||
dq 0x0000000000007F80
|
||||
@@ -0,0 +1,165 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM1": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM2": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM3": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM4": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM5": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM6": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM7": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM8": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM9": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM10": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM11": ["0x8111111111111111", "0x3fff"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
%include "x87cw.mac"
|
||||
|
||||
mov rsp, 0xe000_1000
|
||||
|
||||
finit ; enters x87 state
|
||||
|
||||
; 80-bit mode, round-nearest
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_nearest
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_1]
|
||||
|
||||
; 64-bit mode, round-nearest
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_nearest
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_2]
|
||||
|
||||
; 32-bit mode, round-nearest
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_nearest
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_3]
|
||||
|
||||
; 80-bit mode, round-down
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_down
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_4]
|
||||
|
||||
; 64-bit mode, round-down
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_down
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_5]
|
||||
|
||||
; 32-bit mode, round-down
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_down
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_6]
|
||||
|
||||
; 80-bit mode, round-up
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_up
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_7]
|
||||
|
||||
; 64-bit mode, round-up
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_up
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_8]
|
||||
|
||||
; 32-bit mode, round-up
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_up
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_9]
|
||||
|
||||
; 80-bit mode, round-towards_zero
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_towards_zero
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_10]
|
||||
|
||||
; 64-bit mode, round-towards_zero
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_towards_zero
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_11]
|
||||
|
||||
; 32-bit mode, round-towards_zero
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_towards_zero
|
||||
fld tword [rel .source_1]
|
||||
fabs
|
||||
fstp tword [rel .result_12]
|
||||
|
||||
; Fetch results
|
||||
movups xmm0, [rel .result_1]
|
||||
movups xmm1, [rel .result_2]
|
||||
movups xmm2, [rel .result_3]
|
||||
movups xmm3, [rel .result_4]
|
||||
movups xmm4, [rel .result_5]
|
||||
movups xmm5, [rel .result_6]
|
||||
movups xmm6, [rel .result_7]
|
||||
movups xmm7, [rel .result_8]
|
||||
movups xmm8, [rel .result_9]
|
||||
movups xmm9, [rel .result_10]
|
||||
movups xmm10, [rel .result_11]
|
||||
movups xmm11, [rel .result_12]
|
||||
|
||||
hlt
|
||||
|
||||
; Positive
|
||||
.source_1:
|
||||
dq 0x8111_1111_1111_1111
|
||||
dw 0x3fff
|
||||
|
||||
.result_1:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_2:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_3:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_4:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_5:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_6:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_7:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_8:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_9:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_10:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_11:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_12:
|
||||
dq 0
|
||||
dq 0
|
||||
@@ -0,0 +1,181 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM1": ["0x8111111111111000", "0x3fff"],
|
||||
"XMM2": ["0x8111110000000000", "0x3fff"],
|
||||
"XMM3": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM4": ["0x8111111111111000", "0x3fff"],
|
||||
"XMM5": ["0x8111110000000000", "0x3fff"],
|
||||
"XMM6": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM7": ["0x8111111111111800", "0x3fff"],
|
||||
"XMM8": ["0x8111120000000000", "0x3fff"],
|
||||
"XMM9": ["0x8111111111111111", "0x3fff"],
|
||||
"XMM10": ["0x8111111111111000", "0x3fff"],
|
||||
"XMM11": ["0x8111110000000000", "0x3fff"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
%include "x87cw.mac"
|
||||
|
||||
mov rsp, 0xe000_1000
|
||||
|
||||
finit ; enters x87 state
|
||||
|
||||
; 80-bit mode, round-nearest
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_nearest
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_1]
|
||||
|
||||
; 64-bit mode, round-nearest
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_nearest
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_2]
|
||||
|
||||
; 32-bit mode, round-nearest
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_nearest
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_3]
|
||||
|
||||
; 80-bit mode, round-down
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_down
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_4]
|
||||
|
||||
; 64-bit mode, round-down
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_down
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_5]
|
||||
|
||||
; 32-bit mode, round-down
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_down
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_6]
|
||||
|
||||
; 80-bit mode, round-up
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_up
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_7]
|
||||
|
||||
; 64-bit mode, round-up
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_up
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_8]
|
||||
|
||||
; 32-bit mode, round-up
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_up
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_9]
|
||||
|
||||
; 80-bit mode, round-towards_zero
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_towards_zero
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_10]
|
||||
|
||||
; 64-bit mode, round-towards_zero
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_towards_zero
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_11]
|
||||
|
||||
; 32-bit mode, round-towards_zero
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_towards_zero
|
||||
fld tword [rel .source_1]
|
||||
fld tword [rel .source_zero]
|
||||
faddp
|
||||
fstp tword [rel .result_12]
|
||||
|
||||
; Fetch results
|
||||
movups xmm0, [rel .result_1]
|
||||
movups xmm1, [rel .result_2]
|
||||
movups xmm2, [rel .result_3]
|
||||
movups xmm3, [rel .result_4]
|
||||
movups xmm4, [rel .result_5]
|
||||
movups xmm5, [rel .result_6]
|
||||
movups xmm6, [rel .result_7]
|
||||
movups xmm7, [rel .result_8]
|
||||
movups xmm8, [rel .result_9]
|
||||
movups xmm9, [rel .result_10]
|
||||
movups xmm10, [rel .result_11]
|
||||
movups xmm11, [rel .result_12]
|
||||
|
||||
hlt
|
||||
|
||||
; Positive
|
||||
.source_1:
|
||||
dq 0x8111_1111_1111_1111
|
||||
dw 0x3fff
|
||||
|
||||
.source_zero:
|
||||
dq 0x0
|
||||
dq 0x0
|
||||
|
||||
.result_1:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_2:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_3:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_4:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_5:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_6:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_7:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_8:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_9:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_10:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_11:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_12:
|
||||
dq 0
|
||||
dq 0
|
||||
@@ -0,0 +1,165 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0x86b5441382debef5", "0x3ffe"],
|
||||
"XMM1": ["0x86b5441382debef5", "0x3ffe"],
|
||||
"XMM2": ["0x86b5441382debef5", "0x3ffe"],
|
||||
"XMM3": ["0x86b5441382debef4", "0x3ffe"],
|
||||
"XMM4": ["0x86b5441382debef4", "0x3ffe"],
|
||||
"XMM5": ["0x86b5441382debef4", "0x3ffe"],
|
||||
"XMM6": ["0x86b5441382debef5", "0x3ffe"],
|
||||
"XMM7": ["0x86b5441382debef5", "0x3ffe"],
|
||||
"XMM8": ["0x86b5441382debef5", "0x3ffe"],
|
||||
"XMM9": ["0x86b5441382debef4", "0x3ffe"],
|
||||
"XMM10": ["0x86b5441382debef4", "0x3ffe"],
|
||||
"XMM11": ["0x86b5441382debef4", "0x3ffe"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
%include "x87cw.mac"
|
||||
|
||||
mov rsp, 0xe000_1000
|
||||
|
||||
finit ; enters x87 state
|
||||
|
||||
; 80-bit mode, round-nearest
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_nearest
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_1]
|
||||
|
||||
; 64-bit mode, round-nearest
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_nearest
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_2]
|
||||
|
||||
; 32-bit mode, round-nearest
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_nearest
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_3]
|
||||
|
||||
; 80-bit mode, round-down
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_down
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_4]
|
||||
|
||||
; 64-bit mode, round-down
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_down
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_5]
|
||||
|
||||
; 32-bit mode, round-down
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_down
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_6]
|
||||
|
||||
; 80-bit mode, round-up
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_up
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_7]
|
||||
|
||||
; 64-bit mode, round-up
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_up
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_8]
|
||||
|
||||
; 32-bit mode, round-up
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_up
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_9]
|
||||
|
||||
; 80-bit mode, round-towards_zero
|
||||
set_cw_precision_rounding x87_prec_80, x87_round_towards_zero
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_10]
|
||||
|
||||
; 64-bit mode, round-towards_zero
|
||||
set_cw_precision_rounding x87_prec_64, x87_round_towards_zero
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_11]
|
||||
|
||||
; 32-bit mode, round-towards_zero
|
||||
set_cw_precision_rounding x87_prec_32, x87_round_towards_zero
|
||||
fld tword [rel .source_1]
|
||||
fcos
|
||||
fstp tword [rel .result_12]
|
||||
|
||||
; Fetch results
|
||||
movups xmm0, [rel .result_1]
|
||||
movups xmm1, [rel .result_2]
|
||||
movups xmm2, [rel .result_3]
|
||||
movups xmm3, [rel .result_4]
|
||||
movups xmm4, [rel .result_5]
|
||||
movups xmm5, [rel .result_6]
|
||||
movups xmm6, [rel .result_7]
|
||||
movups xmm7, [rel .result_8]
|
||||
movups xmm8, [rel .result_9]
|
||||
movups xmm9, [rel .result_10]
|
||||
movups xmm10, [rel .result_11]
|
||||
movups xmm11, [rel .result_12]
|
||||
|
||||
hlt
|
||||
|
||||
; Positive
|
||||
.source_1:
|
||||
dq 0x8222_2222_2222_2222
|
||||
dw 0x3fff
|
||||
|
||||
.result_1:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_2:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_3:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_4:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_5:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_6:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_7:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_8:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_9:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_10:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_11:
|
||||
dq 0
|
||||
dq 0
|
||||
|
||||
.result_12:
|
||||
dq 0
|
||||
dq 0
|
||||
Loaded 100 of 174 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user