mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 06:00:16 +02:00
Compare commits
No files matched your search
@@ -59,27 +59,77 @@ jobs:
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
|
||||
- name: IR Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target ir_tests
|
||||
|
||||
- name: IR Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
|
||||
|
||||
- name: Posix Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the posixtest
|
||||
run: cmake --build . --config $BUILD_TYPE --target posix_tests
|
||||
|
||||
- name: Posix Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
|
||||
|
||||
- name: gvisor tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gvisor_tests
|
||||
|
||||
- name: GVisor Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GVisor.log || true
|
||||
|
||||
- name: gcc target tests 64
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_64
|
||||
|
||||
- name: GCC64 Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -103,6 +103,8 @@ if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
|
||||
endif()
|
||||
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter Development)
|
||||
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
|
||||
Vendored
+1
-2
@@ -19,8 +19,7 @@ include(CheckIncludeFileCXX)
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(_M_X86_64 1)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -fno-operator-names -mcx16")
|
||||
set(CMAKE_REQUIRED_DEFINITIONS "-fno-operator-names")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
message(STATUS "Enabling x86-64 JIT")
|
||||
set(ENABLE_JIT 1)
|
||||
endif()
|
||||
|
||||
+12
-5
@@ -77,7 +77,7 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
|
||||
return extF80_roundToInt(lhs, softfloat_round_near_even, false);
|
||||
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
|
||||
@@ -173,19 +173,26 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
operator int16_t() const {
|
||||
return extF80_to_i32(*this, softfloat_round_near_even, false);
|
||||
auto rv = extF80_to_i32(*this, softfloat_roundingMode, false);
|
||||
if (rv > INT16_MAX) {
|
||||
return INT16_MAX;
|
||||
} else if (rv < INT16_MIN) {
|
||||
return INT16_MIN;
|
||||
} else {
|
||||
return rv;
|
||||
}
|
||||
}
|
||||
|
||||
operator int32_t() const {
|
||||
return extF80_to_i32(*this, softfloat_round_near_even, false);
|
||||
return extF80_to_i32(*this, softfloat_roundingMode, false);
|
||||
}
|
||||
|
||||
operator int64_t() const {
|
||||
return extF80_to_i64(*this, softfloat_round_near_even, false);
|
||||
return extF80_to_i64(*this, softfloat_roundingMode, false);
|
||||
}
|
||||
|
||||
operator uint64_t() const {
|
||||
return extF80_to_ui64(*this, softfloat_round_near_even, false);
|
||||
return extF80_to_ui64(*this, softfloat_roundingMode, false);
|
||||
}
|
||||
|
||||
void operator=(const float rhs) {
|
||||
|
||||
+13
-1
@@ -34,7 +34,7 @@ namespace FEXCore::Config {
|
||||
CTX->Config.TSOEnabled = Config != 0;
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_SMC_CHECKS:
|
||||
CTX->Config.SMCChecks = Config != 0;
|
||||
CTX->Config.SMCChecks = static_cast<FEXCore::Config::ConfigSMCChecks>(Config);
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_ABI_LOCAL_FLAGS:
|
||||
CTX->Config.ABILocalFlags = Config != 0;
|
||||
@@ -45,6 +45,12 @@ namespace FEXCore::Config {
|
||||
case FEXCore::Config::CONFIG_VALIDATE_IR_PARSER:
|
||||
CTX->Config.ValidateIRarser = Config != 0;
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_AOTIR_GENERATE:
|
||||
CTX->Config.AOTIRCapture = Config != 0;
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_AOTIR_LOAD:
|
||||
CTX->Config.AOTIRLoad = Config != 0;
|
||||
break;
|
||||
default: LogMan::Msg::A("Unknown configuration option");
|
||||
}
|
||||
}
|
||||
@@ -101,6 +107,12 @@ namespace FEXCore::Config {
|
||||
case FEXCore::Config::CONFIG_VALIDATE_IR_PARSER:
|
||||
return CTX->Config.ValidateIRarser;
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_AOTIR_GENERATE:
|
||||
return CTX->Config.AOTIRCapture;
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_AOTIR_LOAD:
|
||||
return CTX->Config.AOTIRLoad;
|
||||
break;
|
||||
default: LogMan::Msg::A("Unknown configuration option");
|
||||
}
|
||||
|
||||
|
||||
@@ -151,6 +151,21 @@ namespace FEXCore::Context {
|
||||
return CTX->CPUID.RunFunction(Function);
|
||||
}
|
||||
|
||||
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::istream>(const std::string&)> CacheReader) {
|
||||
CTX->AOTIRLoader = CacheReader;
|
||||
}
|
||||
|
||||
bool WriteAOTIR(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
|
||||
return CTX->WriteAOTIRCache(CacheWriter);
|
||||
}
|
||||
|
||||
void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name) {
|
||||
return CTX->AddNamedRegion(Base, Length, Offset, Name);
|
||||
}
|
||||
void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length) {
|
||||
return CTX->RemoveNamedRegion(Base, Length);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
void CompileRIP(FEXCore::Context::Context *CTX, uint64_t RIP) {
|
||||
CTX->CompileRIP(CTX->ParentThread, RIP);
|
||||
|
||||
+43
-4
@@ -6,13 +6,20 @@
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#include <memory>
|
||||
#include <map>
|
||||
#include <unordered_map>
|
||||
#include <set>
|
||||
#include <mutex>
|
||||
#include <istream>
|
||||
#include <ostream>
|
||||
#include <functional>
|
||||
|
||||
namespace FEXCore {
|
||||
class ThunkHandler;
|
||||
@@ -30,6 +37,8 @@ class SyscallHandler;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
namespace Validation {
|
||||
class IRValidation;
|
||||
}
|
||||
@@ -59,10 +68,13 @@ namespace FEXCore::Context {
|
||||
|
||||
bool Is64BitMode {true};
|
||||
bool TSOEnabled {true};
|
||||
bool SMCChecks {false};
|
||||
FEXCore::Config::ConfigSMCChecks SMCChecks {FEXCore::Config::CONFIG_SMC_MMAN};
|
||||
bool ABILocalFlags {false};
|
||||
bool ABINoPF {false};
|
||||
|
||||
bool AOTIRCapture {false};
|
||||
bool AOTIRLoad {false};
|
||||
|
||||
std::string DumpIR;
|
||||
|
||||
// this is for internal use
|
||||
@@ -96,6 +108,28 @@ namespace FEXCore::Context {
|
||||
CustomCPUFactoryType FallbackCPUFactory;
|
||||
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> CustomExitHandler;
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
uint64_t start;
|
||||
uint64_t len;
|
||||
uint64_t crc;
|
||||
IR::IRListView *IR;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
};
|
||||
|
||||
std::function<std::unique_ptr<std::istream>(const std::string&)> AOTIRLoader;
|
||||
std::unordered_map<std::string, std::map<uint64_t, AOTIRCacheEntry>> AOTIRCache;
|
||||
|
||||
struct AddrToFileEntry {
|
||||
uint64_t Start;
|
||||
uint64_t Len;
|
||||
uint64_t Offset;
|
||||
std::string fileid;
|
||||
void *CachedFileEntry;
|
||||
};
|
||||
|
||||
std::map<uint64_t, AddrToFileEntry> AddrToFile;
|
||||
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
#endif
|
||||
@@ -142,11 +176,13 @@ namespace FEXCore::Context {
|
||||
FEXCore::Core::ThreadState *GetThreadState();
|
||||
void LoadEntryList();
|
||||
|
||||
std::tuple<FEXCore::IR::IRListView<true> *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t> GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
std::tuple<FEXCore::IR::IRListView *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t, uint64_t, uint64_t> GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
std::tuple<void *, FEXCore::IR::IRListView<true> *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool> CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
std::tuple<void *, FEXCore::IR::IRListView *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool, uint64_t, uint64_t> CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
uintptr_t CompileBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
bool LoadAOTIRCache(std::istream &stream);
|
||||
bool WriteAOTIRCache(std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
|
||||
// Used for thread creation from syscalls
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
|
||||
@@ -157,6 +193,9 @@ namespace FEXCore::Context {
|
||||
|
||||
std::vector<FEXCore::Core::InternalThreadState*> *const GetThreads() { return &Threads; }
|
||||
|
||||
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
|
||||
|
||||
#if ENABLE_JITSYMBOLS
|
||||
FEXCore::JITSymbols Symbols;
|
||||
#endif
|
||||
@@ -170,7 +209,7 @@ namespace FEXCore::Context {
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void NotifyPause();
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length);
|
||||
|
||||
FEXCore::CodeLoader *LocalLoader{};
|
||||
|
||||
|
||||
+861
-555
File diff suppressed because it is too large.
Load diff
@@ -9,6 +9,22 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t CASAL_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t CASAL_INST = 0x08'E0'FC'00;
|
||||
|
||||
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
|
||||
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
|
||||
|
||||
// Load ops are 4 bits
|
||||
// Acquire and release bits are independent on the instruction
|
||||
constexpr uint32_t ATOMIC_ADD_OP = 0b0000;
|
||||
constexpr uint32_t ATOMIC_CLR_OP = 0b0001;
|
||||
constexpr uint32_t ATOMIC_EOR_OP = 0b0010;
|
||||
constexpr uint32_t ATOMIC_SET_OP = 0b0011;
|
||||
constexpr uint32_t ATOMIC_SMAX_OP = 0b0100;
|
||||
constexpr uint32_t ATOMIC_SMIN_OP = 0b0101;
|
||||
constexpr uint32_t ATOMIC_UMAX_OP = 0b0110;
|
||||
constexpr uint32_t ATOMIC_UMIN_OP = 0b0111;
|
||||
constexpr uint32_t ATOMIC_SWAP_OP = 0b1000;
|
||||
|
||||
bool HandleCASPAL(void *_mcontext, void *_info, uint32_t Instr);
|
||||
bool HandleCASAL(void *_mcontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicMemOp(void *_mcontext, void *_info, uint32_t Instr);
|
||||
}
|
||||
+26
-1
@@ -109,6 +109,31 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h() {
|
||||
return Res;
|
||||
}
|
||||
|
||||
// 2: Cache and TLB information
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h() {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
// returns default values from i7 model 1Ah
|
||||
Res.eax = 0x1 | // Number of iterations needed for all descriptors
|
||||
(0x5A << 8) |
|
||||
(0x03 << 16) |
|
||||
(0x55 << 24);
|
||||
|
||||
Res.ebx = 0xE4 |
|
||||
(0xB2 << 8) |
|
||||
(0xF0 << 16) |
|
||||
(0 << 24);
|
||||
|
||||
Res.ecx = 0; // null descriptors
|
||||
|
||||
Res.edx = 0x2C |
|
||||
(0x21 << 8) |
|
||||
(0xCA << 16) |
|
||||
(0x09 << 24);
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h() {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax = (1 << 2); // Always running APIC
|
||||
@@ -366,7 +391,7 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
CTX = ctx;
|
||||
RegisterFunction(0, std::bind(&CPUIDEmu::Function_0h, this));
|
||||
RegisterFunction(1, std::bind(&CPUIDEmu::Function_01h, this));
|
||||
// 2: Cache and TLB information
|
||||
RegisterFunction(2, std::bind(&CPUIDEmu::Function_02h, this));
|
||||
// 3: Serial Number(previously), now reserved
|
||||
// 4: Deterministic cache parameters for each level
|
||||
// 5: Monitor/mwait
|
||||
|
||||
+7
-1
@@ -3,6 +3,7 @@
|
||||
#include <unordered_map>
|
||||
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -25,8 +26,12 @@ public:
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function) {
|
||||
auto Handler = FunctionHandlers.find(Function);
|
||||
|
||||
if (Handler == FunctionHandlers.end())
|
||||
if (Handler == FunctionHandlers.end()) {
|
||||
#ifndef NDEBUG
|
||||
LogMan::Msg::E("Unhandled CPU ID function, 0x%x", Function);
|
||||
#endif
|
||||
return Function_Reserved();
|
||||
}
|
||||
|
||||
return Handler->second();
|
||||
}
|
||||
@@ -43,6 +48,7 @@ private:
|
||||
// Functions
|
||||
FEXCore::CPUID::FunctionResults Function_0h();
|
||||
FEXCore::CPUID::FunctionResults Function_01h();
|
||||
FEXCore::CPUID::FunctionResults Function_02h();
|
||||
FEXCore::CPUID::FunctionResults Function_06h();
|
||||
FEXCore::CPUID::FunctionResults Function_07h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h();
|
||||
|
||||
@@ -57,9 +57,7 @@ namespace FEXCore {
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Throw::A(CompileThreadData->IRLists.size() == 0, "Compile service must never have IRLists");
|
||||
LogMan::Throw::A(CompileThreadData->RALists.size() == 0, "Compile service must never have RALists");
|
||||
LogMan::Throw::A(CompileThreadData->DebugData.size() == 0, "Compile service must never have DebugData");
|
||||
LogMan::Throw::A(CompileThreadData->LocalIRCache.size() == 0, "Compile service must never have LocalIRCache");
|
||||
|
||||
CompileMutex.unlock();
|
||||
}
|
||||
@@ -128,7 +126,7 @@ namespace FEXCore {
|
||||
// Set our thread state's RIP
|
||||
CompileThreadData->State.State.rip = Item->RIP;
|
||||
|
||||
auto [CodePtr, IRList, DebugData, RAData, Generated] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
|
||||
auto [CodePtr, IRList, DebugData, RAData, Generated, StartAddr, Length] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
|
||||
|
||||
LogMan::Throw::A(Generated == true, "Compile Service doesn't have IR Cache");
|
||||
|
||||
@@ -141,6 +139,8 @@ namespace FEXCore {
|
||||
Item->IRList = IRList;
|
||||
Item->DebugData = DebugData;
|
||||
Item->RAData = RAData;
|
||||
Item->StartAddr = StartAddr;
|
||||
Item->Length = Length;
|
||||
|
||||
GCArray.emplace_back(Item);
|
||||
Item->ServiceWorkDone.NotifyAll();
|
||||
|
||||
+3
-1
@@ -31,9 +31,11 @@ class CompileService final {
|
||||
|
||||
// Outgoing
|
||||
void *CodePtr{};
|
||||
FEXCore::IR::IRListView<true> *IRList{};
|
||||
FEXCore::IR::IRListView *IRList{};
|
||||
FEXCore::IR::RegisterAllocationData *RAData{};
|
||||
FEXCore::Core::DebugData *DebugData{};
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
|
||||
// Communication
|
||||
Event ServiceWorkDone{};
|
||||
|
||||
+329
-59
@@ -24,9 +24,53 @@
|
||||
|
||||
#include <fstream>
|
||||
#include <unistd.h>
|
||||
#include <filesystem>
|
||||
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
|
||||
namespace {
|
||||
// Compression function for Merkle-Damgard construction.
|
||||
// This function is generated using the framework provided.
|
||||
#define mix(h) ({ \
|
||||
(h) ^= (h) >> 23; \
|
||||
(h) *= 0x2127599bf4325c37ULL; \
|
||||
(h) ^= (h) >> 47; })
|
||||
|
||||
static uint64_t fasthash64(const void *buf, size_t len, uint64_t seed)
|
||||
{
|
||||
const uint64_t m = 0x880355f21e6d1965ULL;
|
||||
const uint64_t *pos = (const uint64_t *)buf;
|
||||
const uint64_t *end = pos + (len / 8);
|
||||
const unsigned char *pos2;
|
||||
uint64_t h = seed ^ (len * m);
|
||||
uint64_t v;
|
||||
|
||||
while (pos != end) {
|
||||
v = *pos++;
|
||||
h ^= mix(v);
|
||||
h *= m;
|
||||
}
|
||||
|
||||
pos2 = (const unsigned char*)pos;
|
||||
v = 0;
|
||||
|
||||
switch (len & 7) {
|
||||
case 7: v ^= (uint64_t)pos2[6] << 48;
|
||||
case 6: v ^= (uint64_t)pos2[5] << 40;
|
||||
case 5: v ^= (uint64_t)pos2[4] << 32;
|
||||
case 4: v ^= (uint64_t)pos2[3] << 24;
|
||||
case 3: v ^= (uint64_t)pos2[2] << 16;
|
||||
case 2: v ^= (uint64_t)pos2[1] << 8;
|
||||
case 1: v ^= (uint64_t)pos2[0];
|
||||
h ^= mix(v);
|
||||
h *= m;
|
||||
}
|
||||
|
||||
return mix(h);
|
||||
}
|
||||
#undef mix
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
bool CreateCPUCore(FEXCore::Context::Context *CTX) {
|
||||
// This should be used for generating things that are shared between threads
|
||||
@@ -35,6 +79,8 @@ namespace FEXCore::CPU {
|
||||
}
|
||||
}
|
||||
|
||||
static std::mutex AOTIRCacheLock;
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct ThreadLocalData {
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
@@ -111,7 +157,7 @@ namespace DefaultFallbackCore {
|
||||
void Initialize() override {}
|
||||
bool NeedsOpDispatch() override { return false; }
|
||||
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override {
|
||||
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override {
|
||||
LogMan::Msg::E("Fell back to default code handler at RIP: 0x%lx", ThreadState->State.State.rip);
|
||||
return nullptr;
|
||||
}
|
||||
@@ -155,7 +201,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::AddThreadRIPsToEntryList(FEXCore::Core::InternalThreadState *Thread) {
|
||||
for (auto &IR : Thread->IRLists) {
|
||||
for (auto &IR : Thread->LocalIRCache) {
|
||||
EntryList.insert(IR.first);
|
||||
}
|
||||
}
|
||||
@@ -228,6 +274,14 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
SaveEntryList();
|
||||
|
||||
// AOTIRCache needs manual clear
|
||||
for (auto &Mod: AOTIRCache) {
|
||||
for (auto &Entry: Mod.second) {
|
||||
delete Entry.second.IR;
|
||||
free(Entry.second.RAData);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool Context::InitCore(FEXCore::CodeLoader *Loader) {
|
||||
@@ -250,6 +304,7 @@ namespace FEXCore::Context {
|
||||
memset(NewThreadState.flags, 0, 32);
|
||||
NewThreadState.flags[1] = 1;
|
||||
NewThreadState.flags[9] = 1;
|
||||
NewThreadState.FCW = 0x37F;
|
||||
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(&NewThreadState, 0);
|
||||
|
||||
@@ -411,7 +466,11 @@ namespace FEXCore::Context {
|
||||
CurrentThread = Thread;
|
||||
continue;
|
||||
}
|
||||
StopThread(Thread);
|
||||
if (Thread->State.RunningEvents.Running.load()) {
|
||||
StopThread(Thread);
|
||||
} else {
|
||||
LogMan::Msg::D("Skipping thread %p: Already stopped", Thread);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -422,7 +481,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Thread->State.RunningEvents.Running.load()) {
|
||||
if (Thread->State.RunningEvents.Running.exchange(false)) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::SIGNALEVENT_STOP);
|
||||
tgkill(Thread->State.ThreadManager.PID, Thread->State.ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
@@ -461,9 +520,8 @@ namespace FEXCore::Context {
|
||||
auto IRHandler = [Thread](uint64_t Addr, IR::IREmitter *IR) -> void {
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(IR);
|
||||
Thread->IRLists.try_emplace(Addr, IR->CreateIRCopy());
|
||||
Thread->DebugData.try_emplace(Addr, new Core::DebugData());
|
||||
Thread->RALists.try_emplace(Addr, Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->PullAllocationData() : nullptr);
|
||||
Core::LocalIREntry Entry = {Addr, 0ULL, decltype(Entry.IR)(IR->CreateIRCopy()), decltype(Entry.RAData)(Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->PullAllocationData() : nullptr), decltype(Entry.DebugData)(new Core::DebugData())};
|
||||
Thread->LocalIRCache.insert({Addr, std::move(Entry)});
|
||||
};
|
||||
|
||||
LocalLoader->AddIR(IRHandler);
|
||||
@@ -549,8 +607,8 @@ namespace FEXCore::Context {
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
void Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr, Start, Length);
|
||||
}
|
||||
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache) {
|
||||
@@ -561,13 +619,11 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
if (AlsoClearIRCache) {
|
||||
Thread->IRLists.clear();
|
||||
Thread->RALists.clear();
|
||||
Thread->DebugData.clear();
|
||||
Thread->LocalIRCache.clear();
|
||||
}
|
||||
}
|
||||
|
||||
std::tuple<FEXCore::IR::IRListView<true> *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t> Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
std::tuple<FEXCore::IR::IRListView *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t, uint64_t, uint64_t> Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
uint8_t const *GuestCode{};
|
||||
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
|
||||
|
||||
@@ -581,7 +637,7 @@ namespace FEXCore::Context {
|
||||
LogMan::Msg::E("Had Frontend decoder error");
|
||||
Stop(false /* Ignore Current Thread */);
|
||||
}
|
||||
return { nullptr, nullptr, 0, 0 };
|
||||
return { nullptr, nullptr, 0, 0, 0, 0 };
|
||||
}
|
||||
|
||||
auto CodeBlocks = Thread->FrontendDecoder->GetDecodedBlocks();
|
||||
@@ -614,10 +670,10 @@ namespace FEXCore::Context {
|
||||
TableInfo = Block.DecodedInstructions[i].TableInfo;
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
|
||||
if (Config.SMCChecks) {
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint64_t*>(Block.Entry + BlockInstructionsLength);
|
||||
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(ExistingCodePtr[0], ExistingCodePtr[1], (uintptr_t)ExistingCodePtr, DecodedInfo->InstSize);
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(ExistingCodePtr[0], ExistingCodePtr[1], (uintptr_t)ExistingCodePtr - GuestRIP, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->_CondJump(CodeChanged);
|
||||
|
||||
@@ -626,7 +682,7 @@ namespace FEXCore::Context {
|
||||
Thread->OpDispatcher->SetTrueJumpTarget(InvalidateCodeCond, CodeWasChangedBlock);
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_RemoveCodeEntry(GuestRIP);
|
||||
Thread->OpDispatcher->_RemoveCodeEntry();
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_Constant(Block.Entry + BlockInstructionsLength));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
@@ -661,7 +717,7 @@ namespace FEXCore::Context {
|
||||
if (TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return { nullptr, nullptr, 0, 0 };
|
||||
return { nullptr, nullptr, 0, 0, 0, 0};
|
||||
}
|
||||
else {
|
||||
uint8_t GPRSize = Config.Is64BitMode ? 8 : 4;
|
||||
@@ -756,35 +812,76 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
return {IRList, RAData.release(), TotalInstructions, TotalInstructionsLength};
|
||||
return {IRList, RAData.release(), TotalInstructions, TotalInstructionsLength, Thread->FrontendDecoder->DecodedMinAddress, Thread->FrontendDecoder->DecodedMaxAddress - Thread->FrontendDecoder->DecodedMinAddress };
|
||||
}
|
||||
|
||||
std::tuple<void *, FEXCore::IR::IRListView<true> *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool> Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
FEXCore::IR::IRListView<true> *IRList {};
|
||||
std::tuple<void *, FEXCore::IR::IRListView *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool, uint64_t, uint64_t> Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
|
||||
// Do we already have this in the IR cache?
|
||||
auto IR = Thread->IRLists.find(GuestRIP);
|
||||
auto LocalEntry = Thread->LocalIRCache.find(GuestRIP);
|
||||
|
||||
if (IR != Thread->IRLists.end()) {
|
||||
if (LocalEntry != Thread->LocalIRCache.end()) {
|
||||
// Entry already exists
|
||||
// pull in the data
|
||||
IRList = IR->second.get();
|
||||
DebugData = Thread->DebugData.find(GuestRIP)->second.get();
|
||||
RAData = Thread->RALists.find(GuestRIP)->second.get();
|
||||
IRList = LocalEntry->second.IR.get();
|
||||
DebugData = LocalEntry->second.DebugData.get();
|
||||
RAData = LocalEntry->second.RAData.get();
|
||||
StartAddr = LocalEntry->second.StartAddr;
|
||||
Length = LocalEntry->second.Length;
|
||||
|
||||
GeneratedIR = false;
|
||||
} else {
|
||||
}
|
||||
|
||||
if (IRList == nullptr && Config.AOTIRLoad) {
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
auto Mod = (decltype(AOTIRCache)::value_type::second_type*) file->second.CachedFileEntry;
|
||||
|
||||
if (Mod == nullptr) {
|
||||
file->second.CachedFileEntry = Mod = &AOTIRCache[file->second.fileid];
|
||||
}
|
||||
|
||||
auto AOTEntry = Mod->find(GuestRIP - file->second.Start + file->second.Offset);
|
||||
|
||||
if (AOTEntry != Mod->end()) {
|
||||
// verify hash
|
||||
auto MappedStart = AOTEntry->second.start + file->second.Start - file->second.Offset;
|
||||
auto hash = fasthash64((void*)MappedStart, AOTEntry->second.len, 0);
|
||||
if (hash == AOTEntry->second.crc) {
|
||||
IRList = AOTEntry->second.IR;
|
||||
//LogMan::Msg::D("using %s + %lx -> %lx\n", file->second.fileid.c_str(), AOTEntry->first, GuestRIP);
|
||||
// relocate
|
||||
IRList->GetHeader()->Entry = GuestRIP;
|
||||
|
||||
RAData = AOTEntry->second.RAData;
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
StartAddr = MappedStart;
|
||||
Length = AOTEntry->second.len;
|
||||
|
||||
GeneratedIR = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (IRList == nullptr) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength] = GenerateIR(Thread, GuestRIP);
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP);
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
RAData = RACopy;
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
|
||||
// Initialize metadata
|
||||
DebugData->GuestCodeSize = TotalInstructionsLength;
|
||||
@@ -798,9 +895,135 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
return { Thread->CPUBackend->CompileCode(IRList, DebugData, RAData), IRList, DebugData, RAData, GeneratedIR};
|
||||
return { Thread->CPUBackend->CompileCode(IRList, DebugData, RAData), IRList, DebugData, RAData, GeneratedIR, StartAddr, Length};
|
||||
}
|
||||
|
||||
bool Context::LoadAOTIRCache(std::istream &stream) {
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
uint64_t tag;
|
||||
stream.read((char*)&tag, sizeof(tag));
|
||||
if (!stream || tag != 0xDEADBEEFC0D30002)
|
||||
return false;
|
||||
|
||||
uint64_t ModCount;
|
||||
stream.read((char*)&ModCount, sizeof(ModCount));
|
||||
if (!stream)
|
||||
return false;
|
||||
|
||||
for (int ModIndex = 0; ModIndex < ModCount; ModIndex++) {
|
||||
std::string Module;
|
||||
uint64_t ModSize;
|
||||
stream.read((char*)&ModSize, sizeof(ModSize));
|
||||
if (!stream)
|
||||
return false;
|
||||
|
||||
Module.resize(ModSize);
|
||||
stream.read((char*)&Module[0], Module.size());
|
||||
if (!stream)
|
||||
return false;
|
||||
|
||||
auto &Mod = AOTIRCache[Module];
|
||||
|
||||
uint64_t FnCount;
|
||||
stream.read((char*)&FnCount, sizeof(FnCount));
|
||||
if (!stream)
|
||||
return false;
|
||||
|
||||
LogMan::Msg::D("AOTIR: Module %s has %ld functions", Module.c_str(), FnCount);
|
||||
for (int FnIndex = 0; FnIndex < FnCount; FnIndex++) {
|
||||
uint64_t addr, start, crc, len;
|
||||
stream.read((char*)&addr, sizeof(addr));
|
||||
if (!stream)
|
||||
return false;
|
||||
|
||||
stream.read((char*)&start, sizeof(start));
|
||||
if (!stream)
|
||||
return false;
|
||||
stream.read((char*)&len, sizeof(len));
|
||||
if (!stream)
|
||||
return false;
|
||||
stream.read((char*)&crc, sizeof(crc));
|
||||
if (!stream)
|
||||
return false;
|
||||
auto IR = new IR::IRListView(stream);
|
||||
if (!stream) {
|
||||
delete IR;
|
||||
return false;
|
||||
}
|
||||
uint64_t RASize;
|
||||
stream.read((char*)&RASize, sizeof(RASize));
|
||||
if (!stream) {
|
||||
delete IR;
|
||||
return false;
|
||||
}
|
||||
IR::RegisterAllocationData *RAData = (IR::RegisterAllocationData *)malloc(IR::RegisterAllocationData::Size(RASize));
|
||||
RAData->MapCount = RASize;
|
||||
|
||||
stream.read((char*)&RAData->Map[0], sizeof(RAData->Map[0]) * RASize);
|
||||
|
||||
if (!stream) {
|
||||
delete IR;
|
||||
return false;
|
||||
}
|
||||
stream.read((char*)&RAData->SpillSlotCount, sizeof(RAData->SpillSlotCount));
|
||||
if (!stream) {
|
||||
delete IR;
|
||||
return false;
|
||||
}
|
||||
|
||||
IR->IsShared = true;
|
||||
RAData->IsShared = true;
|
||||
|
||||
Mod.insert({addr, {start, len, crc, IR, RAData}});
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Context::WriteAOTIRCache(std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
|
||||
bool rv = true;
|
||||
|
||||
for (auto AOTModule: AOTIRCache) {
|
||||
if (AOTModule.second.size() == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
auto stream = CacheWriter(AOTModule.first);
|
||||
if (!*stream) {
|
||||
rv = false;
|
||||
}
|
||||
uint64_t tag = 0xDEADBEEFC0D30002;
|
||||
stream->write((char*)&tag, sizeof(tag));
|
||||
|
||||
uint64_t ModCount = 1;
|
||||
stream->write((char*)&ModCount, sizeof(ModCount));
|
||||
auto ModSize = AOTModule.first.size();
|
||||
stream->write((char*)&ModSize, sizeof(ModSize));
|
||||
stream->write((char*)&AOTModule.first[0], ModSize);
|
||||
|
||||
auto FnCount = AOTModule.second.size();
|
||||
stream->write((char*)&FnCount, sizeof(FnCount));
|
||||
|
||||
for (auto entry: AOTModule.second) {
|
||||
stream->write((char*)&entry.first, sizeof(entry.first));
|
||||
stream->write((char*)&entry.second.start, sizeof(entry.second.start));
|
||||
stream->write((char*)&entry.second.len, sizeof(entry.second.len));
|
||||
stream->write((char*)&entry.second.crc, sizeof(entry.second.crc));
|
||||
entry.second.IR->Serialize(*stream);
|
||||
uint64_t RASize = entry.second.RAData->MapCount;
|
||||
stream->write((char*)&RASize, sizeof(RASize));
|
||||
stream->write((char*)&entry.second.RAData->Map[0], sizeof(entry.second.RAData->Map[0]) * RASize);
|
||||
stream->write((char*)&entry.second.RAData->SpillSlotCount, sizeof(entry.second.RAData->SpillSlotCount));
|
||||
}
|
||||
}
|
||||
|
||||
return rv;
|
||||
}
|
||||
|
||||
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
|
||||
// Is the code in the cache?
|
||||
@@ -810,12 +1033,13 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void *CodePtr {};
|
||||
FEXCore::IR::IRListView<true> *IRList {};
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
|
||||
bool DecrementRefCount = false;
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {}, Length {};
|
||||
|
||||
if (Thread->CompileBlockReentrantRefCount != 0) {
|
||||
if (!Thread->CompileService) {
|
||||
@@ -830,6 +1054,8 @@ namespace FEXCore::Context {
|
||||
IRList = WorkItem->IRList;
|
||||
DebugData = WorkItem->DebugData;
|
||||
RAData = WorkItem->RAData;
|
||||
StartAddr = WorkItem->StartAddr;
|
||||
Length = WorkItem->Length;
|
||||
WorkItem->SafeToClear = true;
|
||||
|
||||
// The compile service will always generate IR + DebugData + RAData
|
||||
@@ -839,12 +1065,14 @@ namespace FEXCore::Context {
|
||||
} else {
|
||||
++Thread->CompileBlockReentrantRefCount;
|
||||
DecrementRefCount = true;
|
||||
auto [Code, IR, Data, RA, Generated] = CompileCode(Thread, GuestRIP);
|
||||
auto [Code, IR, Data, RA, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP);
|
||||
CodePtr = Code;
|
||||
IRList = IR;
|
||||
DebugData = Data;
|
||||
RAData = RA;
|
||||
GeneratedIR = Generated;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
}
|
||||
|
||||
LogMan::Throw::A(CodePtr != nullptr, "Failed to compile code %lX", GuestRIP);
|
||||
@@ -864,16 +1092,33 @@ namespace FEXCore::Context {
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
Thread->IRLists.emplace(GuestRIP, IRList);
|
||||
Thread->DebugData.emplace(GuestRIP, DebugData);
|
||||
Thread->RALists.emplace(GuestRIP, RAData);
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), decltype(Entry.RAData)(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
Thread->LocalIRCache.insert({GuestRIP, std::move(Entry)});
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if (Config.AOTIRCapture && RAData) {
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
|
||||
RAData->IsShared = true;
|
||||
IRList->IsShared = true;
|
||||
|
||||
auto hash = fasthash64((void*)StartAddr, Length, 0);
|
||||
|
||||
auto file = AddrToFile.lower_bound(StartAddr);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (file->second.Start <= StartAddr && (file->second.Start + file->second.Len) >= (StartAddr + Length)) {
|
||||
AOTIRCache[file->second.fileid].insert({GuestRIP - file->second.Start + file->second.Offset, {StartAddr - file->second.Start + file->second.Offset, Length, hash, IRList, RAData}});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (DecrementRefCount)
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
|
||||
// Insert to lookup cache
|
||||
AddBlockMapping(Thread, GuestRIP, CodePtr);
|
||||
AddBlockMapping(Thread, GuestRIP, CodePtr, StartAddr, Length);
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
|
||||
@@ -933,10 +1178,22 @@ namespace FEXCore::Context {
|
||||
SignalDelegation->UninstallTLSState(Thread);
|
||||
}
|
||||
|
||||
void FlushCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
|
||||
|
||||
if (Thread->CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_MMAN) {
|
||||
auto lower = Thread->LookupCache->CodePages.lower_bound(Start >> 12);
|
||||
auto upper = Thread->LookupCache->CodePages.upper_bound((Start + Length) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (auto Address: it->second)
|
||||
Context::RemoveCodeEntry(Thread, Address);
|
||||
it->second.clear();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Context::RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
Thread->IRLists.erase(GuestRIP);
|
||||
Thread->RALists.erase(GuestRIP);
|
||||
Thread->DebugData.erase(GuestRIP);
|
||||
Thread->LocalIRCache.erase(GuestRIP);
|
||||
Thread->LookupCache->Erase(GuestRIP);
|
||||
}
|
||||
|
||||
@@ -946,9 +1203,7 @@ namespace FEXCore::Context {
|
||||
Thread->State.State.rip = RIP;
|
||||
|
||||
// Erase the RIP from all the storage backings if it exists
|
||||
Thread->IRLists.erase(RIP);
|
||||
Thread->DebugData.erase(RIP);
|
||||
Thread->LookupCache->Erase(RIP);
|
||||
RemoveCodeEntry(Thread, RIP);
|
||||
|
||||
// We don't care if compilation passes or not
|
||||
CompileBlock(Thread, RIP);
|
||||
@@ -969,12 +1224,12 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
bool Context::GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
auto it = ParentThread->DebugData.find(RIP);
|
||||
if (it == ParentThread->DebugData.end()) {
|
||||
auto it = ParentThread->LocalIRCache.find(RIP);
|
||||
if (it == ParentThread->LocalIRCache.end()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
memcpy(Data, &it->second, sizeof(FEXCore::Core::DebugData));
|
||||
memcpy(Data, it->second.DebugData.get(), sizeof(FEXCore::Core::DebugData));
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -988,21 +1243,6 @@ namespace FEXCore::Context {
|
||||
return true;
|
||||
}
|
||||
|
||||
// XXX:
|
||||
// bool Context::FindIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir) {
|
||||
// auto IR = ParentThread->IRLists.find(RIP);
|
||||
// if (IR == ParentThread->IRLists.end()) {
|
||||
// return false;
|
||||
// }
|
||||
|
||||
// //*ir = &IR->second;
|
||||
// return true;
|
||||
// }
|
||||
|
||||
// void Context::SetIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir) {
|
||||
// //ParentThread->IRLists.try_emplace(RIP, *ir);
|
||||
// }
|
||||
|
||||
FEXCore::Core::ThreadState *Context::GetThreadState() {
|
||||
return &ParentThread->State;
|
||||
}
|
||||
@@ -1013,4 +1253,34 @@ namespace FEXCore::Context {
|
||||
return Result;
|
||||
}
|
||||
|
||||
void Context::AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
// TODO: Support overlapping maps and region splitting
|
||||
auto base_filename = std::filesystem::path(filename).filename().string();
|
||||
|
||||
if (base_filename.size()) {
|
||||
auto filename_hash = fasthash64(filename.c_str(), filename.size(), 0xBAADF00D);
|
||||
|
||||
auto fileid = base_filename + "-" + std::to_string(filename_hash) + "-";
|
||||
|
||||
// append optimization flags to the fileid
|
||||
fileid += (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) ? "S" : "s";
|
||||
fileid += Config.TSOEnabled ? "T" : "t";
|
||||
fileid += Config.ABILocalFlags ? "L" : "l";
|
||||
fileid += Config.ABINoPF ? "p" : "P";
|
||||
|
||||
AddrToFile.insert({ Base, { Base, Size, Offset, fileid, nullptr } });
|
||||
|
||||
if (Config.AOTIRLoad && !AOTIRCache.contains(fileid) && AOTIRLoader) {
|
||||
auto stream = AOTIRLoader(fileid);
|
||||
if (*stream) {
|
||||
LoadAOTIRCache(*stream);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Context::RemoveNamedRegion(uintptr_t Base, uintptr_t Size) {
|
||||
// TODO: Support partial removing
|
||||
AddrToFile.erase(Base);
|
||||
}
|
||||
}
|
||||
+10
-1
@@ -992,6 +992,9 @@ bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
SymbolMinAddress = EntryPoint;
|
||||
}
|
||||
|
||||
DecodedMinAddress = EntryPoint;
|
||||
DecodedMaxAddress = EntryPoint;
|
||||
|
||||
// Entry is a jump target
|
||||
BlocksToDecode.emplace(PC);
|
||||
|
||||
@@ -1015,11 +1018,17 @@ bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
|
||||
if (ErrorDuringDecoding) {
|
||||
LogMan::Msg::D("Couldn't Decode something at 0x%lx, Started at 0x%lx", PC + PCOffset, PC);
|
||||
LogMan::Throw::A(EntryPoint != (RIPToDecode + PCOffset), "Trying to execute invalid code");
|
||||
LogMan::Throw::A(Blocks.size() != 1, "Decode Error in entry block");
|
||||
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
if (ErrorDuringDecoding && Blocks.size() != 1) {
|
||||
ErrorDuringDecoding = false;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
|
||||
DecodedMaxAddress = std::max(DecodedMaxAddress, RIPToDecode + PCOffset + DecodeInst->InstSize);
|
||||
++TotalInstructions;
|
||||
++BlockNumberOfInstructions;
|
||||
++DecodedSize;
|
||||
|
||||
@@ -30,6 +30,9 @@ public:
|
||||
return &Blocks;
|
||||
}
|
||||
|
||||
uint64_t DecodedMinAddress {};
|
||||
uint64_t DecodedMaxAddress {~0ULL};
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
|
||||
|
||||
@@ -22,7 +22,7 @@ public:
|
||||
explicit InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
~InterpreterCore() override;
|
||||
std::string GetName() override { return "Interpreter"; }
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
|
||||
@@ -31,16 +31,9 @@
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
static void InterpreterExecution(FEXCore::Core::InternalThreadState *Thread) {
|
||||
auto IR = Thread->IRLists.find(Thread->State.State.rip)->second.get();
|
||||
auto LocalEntry = Thread->LocalIRCache.find(Thread->State.State.rip);
|
||||
|
||||
FEXCore::Core::DebugData *DebugData = nullptr;
|
||||
|
||||
// DebugData is only used in debug builds
|
||||
#ifndef NDEBUG
|
||||
DebugData = Thread->DebugData.find(Thread->State.State.rip)->second.get();
|
||||
#endif
|
||||
|
||||
InterpreterOps::InterpretIR(Thread, IR, DebugData);
|
||||
InterpreterOps::InterpretIR(Thread, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
|
||||
}
|
||||
|
||||
|
||||
@@ -72,6 +65,18 @@ bool InterpreterCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(_mcontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
_mcontext->pc += 4;
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x: PC: %p Instruction: 0x%08x\n", Op, PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
return false;
|
||||
}
|
||||
@@ -111,7 +116,7 @@ InterpreterCore::~InterpreterCore() {
|
||||
}
|
||||
|
||||
|
||||
void *InterpreterCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void *InterpreterCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
return reinterpret_cast<void*>(InterpreterExecution);
|
||||
}
|
||||
|
||||
|
||||
@@ -409,7 +409,520 @@ static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
|
||||
LogMan::Msg::A("unreachable");
|
||||
}
|
||||
|
||||
void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEXCore::IR::IRListView<true> *CurrentIR, FEXCore::Core::DebugData *DebugData) {
|
||||
template<IR::IROps Op>
|
||||
struct OpHandlers {
|
||||
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
static X80SoftFloat handle4(float src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static X80SoftFloat handle8(double src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CMP> {
|
||||
template<uint32_t Flags>
|
||||
static uint64_t handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
bool eq, lt, nan;
|
||||
uint64_t ResultFlags = 0;
|
||||
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
lt) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
|
||||
nan) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_EQ) &&
|
||||
eq) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
|
||||
}
|
||||
return ResultFlags;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVT> {
|
||||
static float handle4(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static double handle8(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
static int16_t handle2(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static int32_t handle4(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static int64_t handle8(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static int16_t handle2t(X80SoftFloat src) {
|
||||
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
|
||||
if (rv > INT16_MAX) {
|
||||
return INT16_MAX;
|
||||
} else if (rv < INT16_MIN) {
|
||||
return INT16_MIN;
|
||||
} else {
|
||||
return rv;
|
||||
}
|
||||
}
|
||||
|
||||
static int32_t handle4t(X80SoftFloat src) {
|
||||
return extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
}
|
||||
|
||||
static int64_t handle8t(X80SoftFloat src) {
|
||||
return extF80_to_i64(src, softfloat_round_minMag, false);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
static X80SoftFloat handle2(int16_t src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static X80SoftFloat handle4(int32_t src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FRNDINT(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::F2XM1(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FTAN(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SQRT> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FSQRT(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FSIN(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FCOS(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FXTRACT_EXP(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FXTRACT_SIG(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ADD> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FADD(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SUB> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FSUB(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80MUL> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FMUL(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80DIV> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FDIV(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FYL2X(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FATAN(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FREM1(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FREM(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FSCALE(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Sign = 0;
|
||||
|
||||
uint64_t Tmp = Src1;
|
||||
X80SoftFloat Rv;
|
||||
uint8_t *BCD = reinterpret_cast<uint8_t*>(&Rv);
|
||||
memset(BCD, 0, 10);
|
||||
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
if (Tmp == 0) {
|
||||
// Nothing left? Just leave
|
||||
break;
|
||||
}
|
||||
// Extract the lower 100 values
|
||||
uint8_t Digit = Tmp % 100;
|
||||
|
||||
// Now divide it for the next iteration
|
||||
Tmp /= 100;
|
||||
|
||||
uint8_t UpperNibble = Digit / 10;
|
||||
uint8_t LowerNibble = Digit % 10;
|
||||
|
||||
// Now store the BCD
|
||||
BCD[i] = (UpperNibble << 4) | LowerNibble;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
BCD[9] = Negative ? 0x80 : 0;
|
||||
|
||||
return Rv;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src) {
|
||||
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
// Only 0-9 is supported, A-F results in undefined data
|
||||
// | 4 bit | 4 bit |
|
||||
// | 10s place | 1s place |
|
||||
// EG 0x48 = 48
|
||||
// EG 0x4847 = 4847
|
||||
// This gives us an 18digit value encoded in BCD
|
||||
// The last byte lets us know if it negative or not
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
uint8_t Digit = Src1[8 - i];
|
||||
// First shift our last value over
|
||||
BCD *= 100;
|
||||
|
||||
// Add the tens place digit
|
||||
BCD += (Digit >> 4) * 10;
|
||||
|
||||
// Add the ones place digit
|
||||
BCD += Digit & 0xF;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
bool Negative = Src1[9] & 0x80;
|
||||
X80SoftFloat Tmp;
|
||||
|
||||
Tmp = BCD;
|
||||
Tmp.Sign = Negative;
|
||||
return Tmp;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80LOADFCW> {
|
||||
static void handle(uint16_t NewFCW) {
|
||||
|
||||
auto PC = (NewFCW >> 8) & 3;
|
||||
switch(PC) {
|
||||
case 0: extF80_roundingPrecision = 32; break;
|
||||
case 2: extF80_roundingPrecision = 64; break;
|
||||
case 3: extF80_roundingPrecision = 80; break;
|
||||
case 1: LogMan::Msg::A("Invalid x87 precision mode, %d", PC);
|
||||
}
|
||||
|
||||
auto RC = (NewFCW >> 10) & 3;
|
||||
switch(RC) {
|
||||
case 0:
|
||||
softfloat_roundingMode = softfloat_round_near_even;
|
||||
break;
|
||||
case 1:
|
||||
softfloat_roundingMode = softfloat_round_min;
|
||||
break;
|
||||
case 2:
|
||||
softfloat_roundingMode = softfloat_round_max;
|
||||
break;
|
||||
case 3:
|
||||
softfloat_roundingMode = softfloat_round_minMag;
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template<typename R, typename... Args>
|
||||
FallbackInfo GetFallbackInfo(R(*fn)(Args...)) {
|
||||
return {FABI_UNKNOWN, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float)) {
|
||||
return {FABI_F80_F32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double)) {
|
||||
return {FABI_F80_F64, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t)) {
|
||||
return {FABI_F80_I16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t)) {
|
||||
return {FABI_VOID_U16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t)) {
|
||||
return {FABI_F80_I32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I16_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_I64_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_F80_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
*Info = GetFallbackInfo(&OpHandlers<IR::OP_F80LOADFCW>::handle);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&OpHandlers<IR::OP_F80CVTTO>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&OpHandlers<IR::OP_F80CVTTO>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::D("Unhandled size: %d", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVT>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&OpHandlers<IR::OP_F80CVT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&OpHandlers<IR::OP_F80CVT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::D("Unhandled size: %d", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVTINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &OpHandlers<IR::OP_F80CVTINT>::handle2t : &OpHandlers<IR::OP_F80CVTINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &OpHandlers<IR::OP_F80CVTINT>::handle4t : &OpHandlers<IR::OP_F80CVTINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &OpHandlers<IR::OP_F80CVTINT>::handle8t : &OpHandlers<IR::OP_F80CVTINT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::D("Unhandled size: %d", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
decltype(&OpHandlers<IR::OP_F80CMP>::handle<0>) handlers[] = { &OpHandlers<IR::OP_F80CMP>::handle<0>, &OpHandlers<IR::OP_F80CMP>::handle<1>, &OpHandlers<IR::OP_F80CMP>::handle<2>, &OpHandlers<IR::OP_F80CMP>::handle<3>, &OpHandlers<IR::OP_F80CMP>::handle<4>, &OpHandlers<IR::OP_F80CMP>::handle<5>, &OpHandlers<IR::OP_F80CMP>::handle<6>, &OpHandlers<IR::OP_F80CMP>::handle<7> };
|
||||
|
||||
*Info = GetFallbackInfo(handlers[Op->Flags]);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTOINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(&OpHandlers<IR::OP_F80CVTTOINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&OpHandlers<IR::OP_F80CVTTOINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::D("Unhandled size: %d", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = GetFallbackInfo(&OpHandlers<IR::OP_F80##OP>::handle); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_X87_OP(ROUND)
|
||||
COMMON_X87_OP(F2XM1)
|
||||
COMMON_X87_OP(TAN)
|
||||
COMMON_X87_OP(SQRT)
|
||||
COMMON_X87_OP(SIN)
|
||||
COMMON_X87_OP(COS)
|
||||
COMMON_X87_OP(XTRACT_EXP)
|
||||
COMMON_X87_OP(XTRACT_SIG)
|
||||
COMMON_X87_OP(BCDSTORE)
|
||||
COMMON_X87_OP(BCDLOAD)
|
||||
|
||||
// Binary
|
||||
COMMON_X87_OP(ADD)
|
||||
COMMON_X87_OP(SUB)
|
||||
COMMON_X87_OP(MUL)
|
||||
COMMON_X87_OP(DIV)
|
||||
COMMON_X87_OP(FYL2X)
|
||||
COMMON_X87_OP(ATAN)
|
||||
COMMON_X87_OP(FPREM1)
|
||||
COMMON_X87_OP(FPREM)
|
||||
COMMON_X87_OP(SCALE)
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData) {
|
||||
volatile void* stack = alloca(0);
|
||||
|
||||
// Debug data is only passed in debug builds
|
||||
@@ -465,7 +978,8 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEX
|
||||
case IR::OP_VALIDATECODE: {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
|
||||
if (memcmp((void*)Op->CodePtr, &Op->CodeOriginalLow, Op->CodeLength) != 0) {
|
||||
auto CodePtr = CurrentIR->GetHeader()->Entry + Op->Offset;
|
||||
if (memcmp((void*)CodePtr, &Op->CodeOriginalLow, Op->CodeLength) != 0) {
|
||||
GD = 1;
|
||||
} else {
|
||||
GD = 0;
|
||||
@@ -475,7 +989,7 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEX
|
||||
|
||||
case IR::OP_REMOVECODEENTRY: {
|
||||
auto Op = IROp->C<IR::IROp_RemoveCodeEntry>();
|
||||
Thread->CTX->RemoveCodeEntry(Thread, Op->RIP);
|
||||
Thread->CTX->RemoveCodeEntry(Thread, CurrentIR->GetHeader()->Entry);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -576,10 +1090,8 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEX
|
||||
case IR::OP_THUNK: {
|
||||
auto Op = IROp->C<IR::IROp_Thunk>();
|
||||
|
||||
//LogMan::Msg::D("Thunk function: %s, %p, %p\n", Op->ThunkName, Op->ThunkFnPtr, *GetSrc<void**>(Op->Header.Args[0]));
|
||||
|
||||
reinterpret_cast<ThunkedFunction*>(Op->ThunkFnPtr)(*GetSrc<void**>(SSAData, Op->Header.Args[0]));
|
||||
|
||||
auto thunkFn = Thread->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
thunkFn(*GetSrc<void**>(SSAData, Op->Header.Args[0]));
|
||||
break;
|
||||
}
|
||||
case IR::OP_CPUID: {
|
||||
@@ -692,6 +1204,12 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEX
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case IR::OP_ENTRYPOINTOFFSET: {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
GD = CurrentIR->GetHeader()->Entry + Op->Offset;
|
||||
break;
|
||||
}
|
||||
case IR::OP_CONSTANT: {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
GD = Op->Constant;
|
||||
@@ -4206,6 +4724,10 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEX
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80LOADFCW: {
|
||||
OpHandlers<IR::OP_F80LOADFCW>::handle(*GetSrc<uint16_t*>(SSAData, IROp->Args[0]));
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80ADD: {
|
||||
auto Op = IROp->C<IR::IROp_F80Add>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(SSAData, Op->Header.Args[0]);
|
||||
@@ -4321,18 +4843,18 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEX
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
int16_t Tmp = Src;
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
int16_t Tmp = (Op->Truncate? OpHandlers<IR::OP_F80CVTINT>::handle2t : OpHandlers<IR::OP_F80CVTINT>::handle2)(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Tmp = Src;
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
int32_t Tmp = (Op->Truncate? OpHandlers<IR::OP_F80CVTINT>::handle4t : OpHandlers<IR::OP_F80CVTINT>::handle4)(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
int64_t Tmp = Src;
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
int64_t Tmp = (Op->Truncate? OpHandlers<IR::OP_F80CVTINT>::handle8t : OpHandlers<IR::OP_F80CVTINT>::handle8)(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::D("Unhandled size: %d", OpSize);
|
||||
|
||||
@@ -1,19 +1,42 @@
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
template<bool copy>
|
||||
class IRListView;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core{
|
||||
struct DebugData;
|
||||
struct DebugData;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class InterpreterOps {
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEXCore::IR::IRListView<true> *CurrentIR, FEXCore::Core::DebugData *DebugData);
|
||||
};
|
||||
enum FallbackABI {
|
||||
FABI_UNKNOWN,
|
||||
FABI_VOID_U16,
|
||||
FABI_F80_F32,
|
||||
FABI_F80_F64,
|
||||
FABI_F80_I16,
|
||||
FABI_F80_I32,
|
||||
FABI_F32_F80,
|
||||
FABI_F64_F80,
|
||||
FABI_I16_F80,
|
||||
FABI_I32_F80,
|
||||
FABI_I64_F80,
|
||||
FABI_I64_F80_F80,
|
||||
FABI_F80_F80,
|
||||
FABI_F80_F80_F80,
|
||||
};
|
||||
|
||||
struct FallbackInfo {
|
||||
FallbackABI ABI;
|
||||
void *fn;
|
||||
};
|
||||
|
||||
class InterpreterOps {
|
||||
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
|
||||
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
|
||||
};
|
||||
};
|
||||
+14
-5
@@ -51,10 +51,22 @@ DEF_OP(Constant) {
|
||||
LoadConstant(Dst, Op->Constant);
|
||||
}
|
||||
|
||||
DEF_OP(EntrypointOffset) {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
|
||||
auto Constant = IR->GetHeader()->Entry + Op->Offset;
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
LoadConstant(Dst, Constant);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
//nop
|
||||
}
|
||||
|
||||
DEF_OP(InlineEntrypointOffset) {
|
||||
//nop
|
||||
}
|
||||
|
||||
DEF_OP(CycleCounter) {
|
||||
#ifdef DEBUG_CYCLES
|
||||
movz(GetReg<RA_64>(Node), 0);
|
||||
@@ -1043,17 +1055,15 @@ DEF_OP(FCmp) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80Cmp) {
|
||||
LogMan::Msg::D("Unimplemented");
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
void JITCore::RegisterALUHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &JITCore::Op_##x
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
REGISTER_OP(CONSTANT, Constant);
|
||||
REGISTER_OP(ENTRYPOINTOFFSET, EntrypointOffset);
|
||||
REGISTER_OP(INLINECONSTANT, InlineConstant);
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
@@ -1095,7 +1105,6 @@ void JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(FLOAT_TOGPR_U, Float_ToGPR_U);
|
||||
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
|
||||
REGISTER_OP(FCMP, FCmp);
|
||||
REGISTER_OP(F80CMP, F80Cmp);
|
||||
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
+19
-15
@@ -3,6 +3,7 @@
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
@@ -22,9 +23,7 @@ DEF_OP(GuestReturn) {
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// First we must reset the stack
|
||||
if (SpillSlots) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
}
|
||||
ResetStack();
|
||||
|
||||
// Now branch to our signal return helper
|
||||
// This can't be a direct branch since the code needs to live at a constant location
|
||||
@@ -38,9 +37,7 @@ DEF_OP(CallbackReturn) {
|
||||
SpillStaticRegs();
|
||||
|
||||
// First we must reset the stack
|
||||
if (SpillSlots) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
}
|
||||
ResetStack();
|
||||
|
||||
// We can now lower the ref counter again
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(ThreadSharedData.SignalHandlerRefCounterPtr));
|
||||
@@ -64,14 +61,12 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
Label FullLookup;
|
||||
|
||||
if (SpillSlots) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
}
|
||||
ResetStack();
|
||||
|
||||
aarch64::Register RipReg;
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP)) {
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
Literal l_BranchHost{ExitFunctionLinkerAddress};
|
||||
Literal l_BranchGuest{NewRIP};
|
||||
|
||||
@@ -247,7 +242,8 @@ DEF_OP(Thunk) {
|
||||
#if _M_X86_64
|
||||
ERROR_AND_DIE("JIT: OP_THUNK not supported with arm simulator")
|
||||
#else
|
||||
LoadConstant(x2, Op->ThunkFnPtr);
|
||||
auto thunkFn = State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(x2, (uintptr_t)thunkFn);
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
@@ -259,15 +255,23 @@ DEF_OP(Thunk) {
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
uint8_t *NewCode = (uint8_t *)Op->CodePtr;
|
||||
uint8_t *OldCode = (uint8_t *)&Op->CodeOriginalLow;
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
LoadConstant(GetReg<RA_64>(Node), 0);
|
||||
LoadConstant(x0, Op->CodePtr);
|
||||
LoadConstant(x0, IR->GetHeader()->Entry + Op->Offset);
|
||||
LoadConstant(x1, 1);
|
||||
|
||||
|
||||
while (len >= 8)
|
||||
{
|
||||
ldr(x2, MemOperand(x0, idx));
|
||||
LoadConstant(x3, *(uint32_t *)(OldCode + idx));
|
||||
cmp(x2, x3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
len -= 8;
|
||||
idx += 8;
|
||||
}
|
||||
while (len >= 4)
|
||||
{
|
||||
ldr(w2, MemOperand(x0, idx));
|
||||
@@ -306,7 +310,7 @@ DEF_OP(RemoveCodeEntry) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(x0, STATE);
|
||||
LoadConstant(x1, Op->RIP);
|
||||
LoadConstant(x1, IR->GetHeader()->Entry);
|
||||
|
||||
LoadConstant(x2, reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntry));
|
||||
SpillStaticRegs();
|
||||
|
||||
+350
-85
@@ -40,8 +40,260 @@ using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
void JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LogMan::Msg::A("Unhandled IR Op: %s", std::string(Name).c_str());
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LogMan::Msg::A("Unhandled IR Op: %s", std::string(Name).c_str());
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
LoadConstant(x1, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_F32:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
fmov(v0.S(), GetSrc(IROp->Args[0].ID()).S()) ;
|
||||
LoadConstant(x0, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_F64:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
LoadConstant(x0, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_I16:
|
||||
case FABI_F80_I32: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
LoadConstant(x1, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F32_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
fmov(GetDst(Node).S(), v0.S());
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
uxth(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetReg<RA_32>(Node), w0);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(x3, GetSrc(IROp->Args[1].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x4, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(x3, GetSrc(IROp->Args[1].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x4, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
eor(GetDst(Node).V16B(), GetDst(Node).V16B(), GetDst(Node).V16B());
|
||||
ins(GetDst(Node).V2D(), 0, x0);
|
||||
ins(GetDst(Node).V8H(), 4, w1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LogMan::Msg::A("Unhandled IR Fallback abi: %s %d", std::string(Name).c_str(), Info.ABI);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void JITCore::Op_NoOp(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
@@ -296,7 +548,7 @@ bool JITCore::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSig
|
||||
memcpy(guest_uctx->__fpregs_mem._xmm, State->State.State.xmm, sizeof(State->State.State.xmm));
|
||||
|
||||
// FCW store default
|
||||
guest_uctx->__fpregs_mem.fcw = 0x37F;
|
||||
guest_uctx->__fpregs_mem.fcw = State->State.State.FCW;
|
||||
|
||||
// Reconstruct FSW
|
||||
guest_uctx->__fpregs_mem.fsw =
|
||||
@@ -433,7 +685,18 @@ bool JITCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(_mcontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
_mcontext->pc += 4;
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x: PC: %p Instruction: 0x%08x\n", Op, PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
@@ -764,6 +1027,20 @@ bool JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Va
|
||||
}
|
||||
}
|
||||
|
||||
bool JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
*Value = IR->GetHeader()->Entry + Op->Offset;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType JITCore::GetRegClass(uint32_t Node) {
|
||||
return FEXCore::IR::RegisterClassType {GetPhys(RAData, Node).Class};
|
||||
}
|
||||
@@ -781,7 +1058,7 @@ bool JITCore::IsGPR(uint32_t Node) {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
using namespace aarch64;
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
@@ -849,77 +1126,67 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const
|
||||
bind(&RunBlock);
|
||||
}
|
||||
|
||||
if (HeaderOp->ShouldInterpret) {
|
||||
// Make sure RIP is syncronized to the context
|
||||
LoadConstant(x0, HeaderOp->Entry);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::ThreadState, State.rip)));
|
||||
//LogMan::Throw::A(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
|
||||
LoadConstant(x0, ThreadSharedData.InterpreterFallbackHelperAddress);
|
||||
LoadConstant(x1, (uintptr_t)IR);
|
||||
|
||||
// Debug data is only used in debug builds
|
||||
#ifndef NDEBUG
|
||||
LoadConstant(x2, (uintptr_t)DebugData);
|
||||
#endif
|
||||
br(x0);
|
||||
} else {
|
||||
//LogMan::Throw::A(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
if (SpillSlots) {
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
sub(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
sub(sp, sp, x0);
|
||||
}
|
||||
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
{
|
||||
uint32_t Node = IR->GetID(BlockNode);
|
||||
auto IsTarget = JumpTargets.find(Node);
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
}
|
||||
|
||||
// if there's a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
|
||||
{
|
||||
b(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
bind(&IsTarget->second);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()), 0, IR->GetID(BlockNode)});
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
uint32_t ID = IR->GetID(CodeNode);
|
||||
|
||||
// Execute handler
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
(this->*Handler)(IROp, ID);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.back().HostCodeSize = Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()) - DebugData->Subblocks.back().HostCodeStart;
|
||||
}
|
||||
}
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
if (PendingTargetLabel)
|
||||
{
|
||||
b(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
}
|
||||
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
{
|
||||
uint32_t Node = IR->GetID(BlockNode);
|
||||
auto IsTarget = JumpTargets.find(Node);
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
}
|
||||
|
||||
// if there's a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
|
||||
{
|
||||
b(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
bind(&IsTarget->second);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()), 0, IR->GetID(BlockNode)});
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
uint32_t ID = IR->GetID(CodeNode);
|
||||
|
||||
// Execute handler
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
(this->*Handler)(IROp, ID);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.back().HostCodeSize = Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()) - DebugData->Subblocks.back().HostCodeStart;
|
||||
}
|
||||
}
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
if (PendingTargetLabel)
|
||||
{
|
||||
b(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
FinalizeCode();
|
||||
|
||||
auto CodeEnd = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
@@ -1099,7 +1366,6 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Literal l_VirtualMemory {VirtualMemorySize};
|
||||
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
|
||||
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
|
||||
Literal l_Interpreter {reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR)};
|
||||
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
|
||||
|
||||
uintptr_t CompileBlockPtr{};
|
||||
@@ -1280,19 +1546,6 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
hlt(0);
|
||||
}
|
||||
|
||||
{
|
||||
Label InterpreterFallback{};
|
||||
bind(&InterpreterFallback);
|
||||
ThreadSharedData.InterpreterFallbackHelperAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
SpillStaticRegs();
|
||||
mov(x0, STATE);
|
||||
ldr(x3, &l_Interpreter);
|
||||
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
SpillStaticRegs();
|
||||
@@ -1367,7 +1620,6 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
place(&l_VirtualMemory);
|
||||
place(&l_PagePtr);
|
||||
place(&l_CTX);
|
||||
place(&l_Interpreter);
|
||||
place(&l_Sleep);
|
||||
place(&l_CompileBlock);
|
||||
place(&l_ExitFunctionLink);
|
||||
@@ -1450,6 +1702,19 @@ void JITCore::PopDynamicRegsAndLR() {
|
||||
add(sp, sp, SPOffset);
|
||||
}
|
||||
|
||||
void JITCore::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
return;
|
||||
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::CPU::CPUBackend *CreateJITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return new JITCore(ctx, Thread, JITCore::AllocateNewCodeBuffer(JITCore::INITIAL_CODE_SIZE), CompileThread);
|
||||
}
|
||||
|
||||
@@ -78,7 +78,7 @@ public:
|
||||
|
||||
~JITCore() override;
|
||||
std::string GetName() override { return "JIT"; }
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -100,7 +100,7 @@ private:
|
||||
Label *PendingTargetLabel;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *State;
|
||||
FEXCore::IR::IRListView<true> const *IR;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
|
||||
std::map<IR::OrderedNodeWrapper::NodeOffsetType, aarch64::Label> JumpTargets;
|
||||
|
||||
@@ -152,6 +152,7 @@ private:
|
||||
MemOperand GenerateMemOperand(uint8_t AccessSize, aarch64::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr);
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value);
|
||||
|
||||
struct LiveRange {
|
||||
uint32_t Begin;
|
||||
@@ -223,8 +224,6 @@ private:
|
||||
/** @} */
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t InterpreterFallbackHelperAddress{};
|
||||
|
||||
uint64_t SignalReturnInstruction{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
@@ -242,6 +241,8 @@ private:
|
||||
void PushDynamicRegsAndLR();
|
||||
void PopDynamicRegsAndLR();
|
||||
|
||||
void ResetStack();
|
||||
|
||||
using OpHandler = void (JITCore::*)(FEXCore::IR::IROp_Header *IROp, uint32_t Node);
|
||||
std::array<OpHandler, FEXCore::IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
void RegisterALUHandlers();
|
||||
@@ -265,7 +266,9 @@ private:
|
||||
///< ALU Ops
|
||||
DEF_OP(TruncElementPair);
|
||||
DEF_OP(Constant);
|
||||
DEF_OP(EntrypointOffset);
|
||||
DEF_OP(InlineConstant);
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
@@ -309,7 +312,6 @@ private:
|
||||
DEF_OP(Float_ToGPR_U);
|
||||
DEF_OP(Float_ToGPR_S);
|
||||
DEF_OP(FCmp);
|
||||
DEF_OP(F80Cmp);
|
||||
|
||||
///< Atomic ops
|
||||
DEF_OP(CASPair);
|
||||
|
||||
@@ -41,9 +41,7 @@ DEF_OP(Break) {
|
||||
break;
|
||||
}
|
||||
case 6: { // INT3
|
||||
if (SpillSlots) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
}
|
||||
ResetStack();
|
||||
|
||||
LoadConstant(TMP1, ThreadPauseHandlerAddressSpillSRA);
|
||||
br(TMP1);
|
||||
|
||||
@@ -29,18 +29,21 @@ DEF_OP(CreateElementPair) {
|
||||
std::pair<aarch64::Register, aarch64::Register> Dst;
|
||||
aarch64::Register RegFirst;
|
||||
aarch64::Register RegSecond;
|
||||
aarch64::Register RegTmp;
|
||||
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetReg<RA_32>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Header.Args[1].ID());
|
||||
RegTmp = w0;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
RegTmp = x0;
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::A("Unknown Size"); break;
|
||||
@@ -53,7 +56,9 @@ DEF_OP(CreateElementPair) {
|
||||
mov(Dst.second, RegSecond);
|
||||
mov(Dst.first, RegFirst);
|
||||
} else {
|
||||
LogMan::Msg::A("Unhandled CreateElementPair");
|
||||
mov(RegTmp, RegFirst);
|
||||
mov(Dst.second, RegSecond);
|
||||
mov(Dst.first, RegTmp);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+36
-23
@@ -23,17 +23,28 @@ DEF_OP(Constant) {
|
||||
mov(GetDst<RA_64>(Node), Op->Constant);
|
||||
}
|
||||
|
||||
DEF_OP(EntrypointOffset) {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
|
||||
auto Constant = IR->GetHeader()->Entry + Op->Offset;
|
||||
mov(GetDst<RA_64>(Node), Constant);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
//nop
|
||||
}
|
||||
|
||||
DEF_OP(InlineEntrypointOffset) {
|
||||
//nop
|
||||
}
|
||||
|
||||
DEF_OP(CycleCounter) {
|
||||
#ifdef DEBUG_CYCLES
|
||||
mov (GetDst<RA_64>(Node), 0);
|
||||
#else
|
||||
rdtsc();
|
||||
shl(rdx, 32);
|
||||
or(rax, rdx);
|
||||
or_(rax, rdx);
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
#endif
|
||||
}
|
||||
@@ -373,9 +384,9 @@ DEF_OP(Or) {
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Header.Args[1], &Const)) {
|
||||
or (rax, Const);
|
||||
or_(rax, Const);
|
||||
} else {
|
||||
or (rax, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
or_(rax, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
}
|
||||
mov(Dst, rax);
|
||||
}
|
||||
@@ -386,9 +397,9 @@ DEF_OP(And) {
|
||||
mov(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Header.Args[1], &Const)) {
|
||||
and (rax, Const);
|
||||
and_(rax, Const);
|
||||
} else {
|
||||
and(rax, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
and_(rax, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
}
|
||||
mov(Dst, rax);
|
||||
}
|
||||
@@ -399,9 +410,9 @@ DEF_OP(Xor) {
|
||||
mov(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Header.Args[1], &Const)) {
|
||||
xor(rax, Const);
|
||||
xor_(rax, Const);
|
||||
} else {
|
||||
xor(rax, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
xor_(rax, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
}
|
||||
mov(Dst, rax);
|
||||
}
|
||||
@@ -428,7 +439,7 @@ DEF_OP(Lshl) {
|
||||
};
|
||||
} else {
|
||||
mov(rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
and(rcx, Mask);
|
||||
and_(rcx, Mask);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
@@ -476,7 +487,7 @@ DEF_OP(Lshr) {
|
||||
|
||||
} else {
|
||||
mov (rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
and(rcx, Mask);
|
||||
and_(rcx, Mask);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -534,7 +545,7 @@ DEF_OP(Ashr) {
|
||||
|
||||
} else {
|
||||
mov (rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
and(rcx, Mask);
|
||||
and_(rcx, Mask);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
movsx(rax, GetSrc<RA_8>(Op->Header.Args[0].ID()));
|
||||
@@ -583,7 +594,7 @@ DEF_OP(Ror) {
|
||||
}
|
||||
} else {
|
||||
mov (rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
and(rcx, Mask);
|
||||
and_(rcx, Mask);
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
mov(eax, GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
@@ -790,10 +801,10 @@ DEF_OP(FindLSB) {
|
||||
bsf(rcx, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(rax, 0x40);
|
||||
cmovz(rcx, rax);
|
||||
xor(rax, rax);
|
||||
xor_(rax, rax);
|
||||
cmp(GetSrc<RA_64>(Op->Header.Args[0].ID()), 1);
|
||||
sbb(rax, rax);
|
||||
or(rax, rcx);
|
||||
or_(rax, rcx);
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
|
||||
@@ -870,7 +881,7 @@ DEF_OP(CountLeadingZeroes) {
|
||||
Label Skip;
|
||||
je(Skip);
|
||||
bsr(ax, GetSrc<RA_16>(Op->Header.Args[0].ID()));
|
||||
xor(ax, 0xF);
|
||||
xor_(ax, 0xF);
|
||||
movzx(eax, ax);
|
||||
L(Skip);
|
||||
mov(GetDst<RA_32>(Node), eax);
|
||||
@@ -882,7 +893,7 @@ DEF_OP(CountLeadingZeroes) {
|
||||
Label Skip;
|
||||
je(Skip);
|
||||
bsr(eax, GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
xor(eax, 0x1F);
|
||||
xor_(eax, 0x1F);
|
||||
L(Skip);
|
||||
mov(GetDst<RA_32>(Node), eax);
|
||||
break;
|
||||
@@ -893,7 +904,7 @@ DEF_OP(CountLeadingZeroes) {
|
||||
Label Skip;
|
||||
je(Skip);
|
||||
bsr(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
xor(rax, 0x3F);
|
||||
xor_(rax, 0x3F);
|
||||
L(Skip);
|
||||
mov(GetDst<RA_64>(Node), rax);
|
||||
break;
|
||||
@@ -939,15 +950,15 @@ DEF_OP(Bfi) {
|
||||
mov(Dst, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
mov(TMP2, DestMask);
|
||||
and(Dst, TMP2);
|
||||
and_(Dst, TMP2);
|
||||
mov(TMP2, SourceMask);
|
||||
and(TMP1, TMP2);
|
||||
and_(TMP1, TMP2);
|
||||
shl(TMP1, Op->lsb);
|
||||
or_(Dst, TMP1);
|
||||
|
||||
if (OpSize != 8) {
|
||||
mov(rcx, uint64_t((1ULL << (OpSize * 8)) - 1));
|
||||
and(Dst, rcx);
|
||||
and_(Dst, rcx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -987,7 +998,7 @@ DEF_OP(Bfe) {
|
||||
|
||||
if (Op->Width != 64) {
|
||||
mov(rcx, uint64_t((1ULL << Op->Width) - 1));
|
||||
and(Dst, rcx);
|
||||
and_(Dst, rcx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1136,21 +1147,21 @@ DEF_OP(FCmp) {
|
||||
mov(rcx, 0);
|
||||
setb(cl);
|
||||
shl(rcx, IR::FCMP_FLAG_LT);
|
||||
or(rdx, rcx);
|
||||
or_(rdx, rcx);
|
||||
}
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED)) {
|
||||
sahf();
|
||||
mov(rcx, 0);
|
||||
setp(cl);
|
||||
shl(rcx, IR::FCMP_FLAG_UNORDERED);
|
||||
or(rdx, rcx);
|
||||
or_(rdx, rcx);
|
||||
}
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ)) {
|
||||
sahf();
|
||||
mov(rcx, 0);
|
||||
setz(cl);
|
||||
shl(rcx, IR::FCMP_FLAG_EQ);
|
||||
or(rdx, rcx);
|
||||
or_(rdx, rcx);
|
||||
}
|
||||
mov (GetDst<RA_64>(Node), rdx);
|
||||
}
|
||||
@@ -1161,7 +1172,9 @@ void JITCore::RegisterALUHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &JITCore::Op_##x
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
REGISTER_OP(CONSTANT, Constant);
|
||||
REGISTER_OP(ENTRYPOINTOFFSET, EntrypointOffset);
|
||||
REGISTER_OP(INLINECONSTANT, InlineConstant);
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
|
||||
+24
-24
@@ -155,16 +155,16 @@ DEF_OP(AtomicAnd) {
|
||||
lock();
|
||||
switch (Op->Size) {
|
||||
case 1:
|
||||
and(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
and_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 2:
|
||||
and(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
and_(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 4:
|
||||
and(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 8:
|
||||
and(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
and_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LogMan::Msg::A("Unhandled AtomicAdd size: %d", Op->Size);
|
||||
}
|
||||
@@ -177,16 +177,16 @@ DEF_OP(AtomicOr) {
|
||||
lock();
|
||||
switch (Op->Size) {
|
||||
case 1:
|
||||
or(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
or_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 2:
|
||||
or(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
or_(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 4:
|
||||
or(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
or_(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 8:
|
||||
or(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
or_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LogMan::Msg::A("Unhandled AtomicAdd size: %d", Op->Size);
|
||||
}
|
||||
@@ -199,16 +199,16 @@ DEF_OP(AtomicXor) {
|
||||
lock();
|
||||
switch (Op->Size) {
|
||||
case 1:
|
||||
xor(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
xor_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 2:
|
||||
xor(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
xor_(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 4:
|
||||
xor(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
xor_(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
case 8:
|
||||
xor(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
xor_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
default: LogMan::Msg::A("Unhandled AtomicAdd size: %d", Op->Size);
|
||||
}
|
||||
@@ -329,7 +329,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt8(), TMP1.cvt8());
|
||||
mov(TMP3.cvt8(), TMP1.cvt8());
|
||||
and(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
|
||||
@@ -345,7 +345,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt16(), TMP1.cvt16());
|
||||
mov(TMP3.cvt16(), TMP1.cvt16());
|
||||
and(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
|
||||
@@ -362,7 +362,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt32(), TMP1.cvt32());
|
||||
mov(TMP3.cvt32(), TMP1.cvt32());
|
||||
and(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
|
||||
@@ -379,7 +379,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt64(), TMP1.cvt64());
|
||||
mov(TMP3.cvt64(), TMP1.cvt64());
|
||||
and(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
and_(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
|
||||
@@ -406,7 +406,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt8(), TMP1.cvt8());
|
||||
mov(TMP3.cvt8(), TMP1.cvt8());
|
||||
or(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
or_(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
|
||||
@@ -422,7 +422,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt16(), TMP1.cvt16());
|
||||
mov(TMP3.cvt16(), TMP1.cvt16());
|
||||
or(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
or_(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
|
||||
@@ -439,7 +439,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt32(), TMP1.cvt32());
|
||||
mov(TMP3.cvt32(), TMP1.cvt32());
|
||||
or(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
or_(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
|
||||
@@ -456,7 +456,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt64(), TMP1.cvt64());
|
||||
mov(TMP3.cvt64(), TMP1.cvt64());
|
||||
or(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
or_(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
|
||||
@@ -483,7 +483,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt8(), TMP1.cvt8());
|
||||
mov(TMP3.cvt8(), TMP1.cvt8());
|
||||
xor(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
xor_(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
|
||||
@@ -499,7 +499,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt16(), TMP1.cvt16());
|
||||
mov(TMP3.cvt16(), TMP1.cvt16());
|
||||
xor(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
xor_(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
|
||||
@@ -516,7 +516,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt32(), TMP1.cvt32());
|
||||
mov(TMP3.cvt32(), TMP1.cvt32());
|
||||
xor(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
xor_(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
|
||||
@@ -533,7 +533,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
L(Loop);
|
||||
mov(TMP2.cvt64(), TMP1.cvt64());
|
||||
mov(TMP3.cvt64(), TMP1.cvt64());
|
||||
xor(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
xor_(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
|
||||
// Updates RAX with the value from memory
|
||||
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
@@ -66,7 +67,7 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP)) {
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
Label l_BranchHost;
|
||||
Label l_BranchGuest;
|
||||
|
||||
@@ -226,8 +227,10 @@ DEF_OP(Thunk) {
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
mov(rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
|
||||
mov(rax, reinterpret_cast<uintptr_t>(Op->ThunkFnPtr));
|
||||
mov(rax, reinterpret_cast<uintptr_t>(thunkFn));
|
||||
call(rax);
|
||||
|
||||
if (NumPush & 1)
|
||||
@@ -244,7 +247,7 @@ DEF_OP(ValidateCode) {
|
||||
int idx = 0;
|
||||
|
||||
xor_(GetDst<RA_64>(Node), GetDst<RA_64>(Node));
|
||||
mov(rax, Op->CodePtr);
|
||||
mov(rax, IR->GetHeader()->Entry + Op->Offset);
|
||||
mov(rbx, 1);
|
||||
while (len >= 4) {
|
||||
cmp(dword[rax + idx], *(uint32_t*)(OldCode + idx));
|
||||
@@ -279,7 +282,7 @@ DEF_OP(RemoveCodeEntry) {
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
mov(rdi, STATE);
|
||||
mov(rax, Op->RIP); // imm64 move
|
||||
mov(rax, IR->GetHeader()->Entry); // imm64 move
|
||||
mov(rsi, rax);
|
||||
|
||||
|
||||
|
||||
@@ -9,7 +9,7 @@ DEF_OP(GetHostFlag) {
|
||||
|
||||
mov(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
shr(rax, Op->Flag);
|
||||
and(rax, 1);
|
||||
and_(rax, 1);
|
||||
mov(GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
|
||||
|
||||
+360
-142
@@ -196,7 +196,7 @@ bool JITCore::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSig
|
||||
memcpy(guest_uctx->__fpregs_mem._xmm, ThreadState->State.State.xmm, sizeof(ThreadState->State.State.xmm));
|
||||
|
||||
// FCW store default
|
||||
guest_uctx->__fpregs_mem.fcw = 0x37F;
|
||||
guest_uctx->__fpregs_mem.fcw = ThreadState->State.State.FCW;
|
||||
|
||||
// Reconstruct FSW
|
||||
guest_uctx->__fpregs_mem.fsw =
|
||||
@@ -339,9 +339,239 @@ bool JITCore::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
return false;
|
||||
}
|
||||
|
||||
void JITCore::PushRegs() {
|
||||
for (auto &Xmm : RAXMM_x) {
|
||||
sub(rsp, 16);
|
||||
movaps(ptr[rsp], Xmm);
|
||||
}
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
push(Reg);
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
}
|
||||
|
||||
void JITCore::PopRegs() {
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
for (uint32_t i = RAXMM_x.size(); i > 0; --i) {
|
||||
movaps(RAXMM_x[i - 1], ptr[rsp]);
|
||||
add(rsp, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LogMan::Msg::A("Unhandled IR Op: %s", std::string(Name).c_str());
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LogMan::Msg::A("Unhandled IR Op: %s", std::string(Name).c_str());
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16: {
|
||||
PushRegs();
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
break;
|
||||
}
|
||||
case FABI_F80_F32:{
|
||||
PushRegs();
|
||||
|
||||
movss(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
pxor(GetDst(Node), GetDst(Node));
|
||||
movq(GetDst(Node), rax);
|
||||
pinsrw(GetDst(Node), edx, 4);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_F64:{
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
pxor(GetDst(Node), GetDst(Node));
|
||||
movq(GetDst(Node), rax);
|
||||
pinsrw(GetDst(Node), edx, 4);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_I16:
|
||||
case FABI_F80_I32: {
|
||||
PushRegs();
|
||||
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
pxor(GetDst(Node), GetDst(Node));
|
||||
movq(GetDst(Node), rax);
|
||||
pinsrw(GetDst(Node), edx, 4);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F32_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
movss(GetDst(Node), xmm0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
movsd(GetDst(Node), xmm0);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
movzx(GetDst<RA_64>(Node), ax);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
mov(GetDst<RA_32>(Node), eax);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
mov(GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
mov(GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
pxor(GetDst(Node), GetDst(Node));
|
||||
movq(GetDst(Node), rax);
|
||||
pinsrw(GetDst(Node), edx, 4);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
mov(rax, (uintptr_t)Info.fn);
|
||||
|
||||
call(rax);
|
||||
|
||||
PopRegs();
|
||||
|
||||
pxor(GetDst(Node), GetDst(Node));
|
||||
movq(GetDst(Node), rax);
|
||||
pinsrw(GetDst(Node), edx, 4);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LogMan::Msg::A("Unhandled IR Fallback abi: %s %d", std::string(Name).c_str(), Info.ABI);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void JITCore::Op_NoOp(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
@@ -575,6 +805,20 @@ bool JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Va
|
||||
}
|
||||
}
|
||||
|
||||
bool JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
*Value = IR->GetHeader()->Entry + Op->Offset;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
std::tuple<JITCore::SetCC, JITCore::CMovCC, JITCore::JCC> JITCore::GetCC(IR::CondClassType cond) {
|
||||
switch (cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return { &CodeGenerator::sete , &CodeGenerator::cmove , &CodeGenerator::je };
|
||||
@@ -608,7 +852,7 @@ std::tuple<JITCore::SetCC, JITCore::CMovCC, JITCore::JCC> JITCore::GetCC(IR::Con
|
||||
return { &CodeGenerator::sete , &CodeGenerator::cmove , &CodeGenerator::je };
|
||||
}
|
||||
|
||||
void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
@@ -643,147 +887,132 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const
|
||||
L(RunBlock);
|
||||
}
|
||||
|
||||
if (HeaderOp->ShouldInterpret) {
|
||||
mov(rax, HeaderOp->Entry);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CPUState, rip)], rax);
|
||||
mov(rsi, (uint64_t)IR);
|
||||
LogMan::Throw::A(RAData != nullptr, "Needs RA");
|
||||
|
||||
// Debug data is only used in debug builds
|
||||
#ifndef NDEBUG
|
||||
mov(rdx, (uint64_t)DebugData);
|
||||
#endif
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
mov(rax, (uintptr_t)ThreadSharedData.InterpreterFallbackHelperAddress);
|
||||
jmp(rax);
|
||||
} else {
|
||||
LogMan::Throw::A(RAData != nullptr, "Needs RA");
|
||||
if (SpillSlots) {
|
||||
sub(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
#ifdef BLOCKSTATS
|
||||
BlockSamplingData::BlockData *SamplingData = CTX->BlockData->GetBlockData(HeaderOp->Entry);
|
||||
if (GetSamplingData) {
|
||||
mov(rcx, reinterpret_cast<uintptr_t>(SamplingData));
|
||||
rdtsc();
|
||||
shl(rdx, 32);
|
||||
or_(rax, rdx);
|
||||
mov(qword [rcx + offsetof(BlockSamplingData::BlockData, Start)], rax);
|
||||
}
|
||||
|
||||
if (SpillSlots) {
|
||||
sub(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
BlockSamplingData::BlockData *SamplingData = CTX->BlockData->GetBlockData(HeaderOp->Entry);
|
||||
auto ExitBlock = [&]() {
|
||||
if (GetSamplingData) {
|
||||
mov(rcx, reinterpret_cast<uintptr_t>(SamplingData));
|
||||
// Get time
|
||||
rdtsc();
|
||||
shl(rdx, 32);
|
||||
or(rax, rdx);
|
||||
mov(qword [rcx + offsetof(BlockSamplingData::BlockData, Start)], rax);
|
||||
or_(rax, rdx);
|
||||
|
||||
// Calculate time spent in block
|
||||
mov(rdx, qword [rcx + offsetof(BlockSamplingData::BlockData, Start)]);
|
||||
sub(rax, rdx);
|
||||
|
||||
// Add time to total time
|
||||
add(qword [rcx + offsetof(BlockSamplingData::BlockData, TotalTime)], rax);
|
||||
|
||||
// Increment call count
|
||||
inc(qword [rcx + offsetof(BlockSamplingData::BlockData, TotalCalls)]);
|
||||
|
||||
// Calculate min
|
||||
mov(rdx, qword [rcx + offsetof(BlockSamplingData::BlockData, Min)]);
|
||||
cmp(rdx, rax);
|
||||
cmova(rdx, rax);
|
||||
mov(qword [rcx + offsetof(BlockSamplingData::BlockData, Min)], rdx);
|
||||
|
||||
// Calculate max
|
||||
mov(rdx, qword [rcx + offsetof(BlockSamplingData::BlockData, Max)]);
|
||||
cmp(rdx, rax);
|
||||
cmovb(rdx, rax);
|
||||
mov(qword [rcx + offsetof(BlockSamplingData::BlockData, Max)], rdx);
|
||||
}
|
||||
};
|
||||
#endif
|
||||
|
||||
auto ExitBlock = [&]() {
|
||||
if (GetSamplingData) {
|
||||
mov(rcx, reinterpret_cast<uintptr_t>(SamplingData));
|
||||
// Get time
|
||||
rdtsc();
|
||||
shl(rdx, 32);
|
||||
or(rax, rdx);
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
// Calculate time spent in block
|
||||
mov(rdx, qword [rcx + offsetof(BlockSamplingData::BlockData, Start)]);
|
||||
sub(rax, rdx);
|
||||
|
||||
// Add time to total time
|
||||
add(qword [rcx + offsetof(BlockSamplingData::BlockData, TotalTime)], rax);
|
||||
|
||||
// Increment call count
|
||||
inc(qword [rcx + offsetof(BlockSamplingData::BlockData, TotalCalls)]);
|
||||
|
||||
// Calculate min
|
||||
mov(rdx, qword [rcx + offsetof(BlockSamplingData::BlockData, Min)]);
|
||||
cmp(rdx, rax);
|
||||
cmova(rdx, rax);
|
||||
mov(qword [rcx + offsetof(BlockSamplingData::BlockData, Min)], rdx);
|
||||
|
||||
// Calculate max
|
||||
mov(rdx, qword [rcx + offsetof(BlockSamplingData::BlockData, Max)]);
|
||||
cmp(rdx, rax);
|
||||
cmovb(rdx, rax);
|
||||
mov(qword [rcx + offsetof(BlockSamplingData::BlockData, Max)], rdx);
|
||||
}
|
||||
};
|
||||
#endif
|
||||
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
{
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
uint32_t Node = IR->GetID(BlockNode);
|
||||
auto IsTarget = JumpTargets.find(Node);
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
}
|
||||
|
||||
// if there is a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
|
||||
{
|
||||
jmp(*PendingTargetLabel, T_NEAR);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
L(IsTarget->second);
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
#ifdef DEBUG_RA
|
||||
if (IROp->Op != IR::OP_BEGINBLOCK &&
|
||||
IROp->Op != IR::OP_CONDJUMP &&
|
||||
IROp->Op != IR::OP_JUMP) {
|
||||
std::stringstream Inst;
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
|
||||
if (IROp->HasDest) {
|
||||
uint64_t PhysReg = RAPass->GetNodeRegister(Node);
|
||||
if (PhysReg >= GPRPairBase)
|
||||
Inst << "\tPair" << GetPhys(Node) << " = " << Name << " ";
|
||||
else if (PhysReg >= XMMBase)
|
||||
Inst << "\tXMM" << GetPhys(Node) << " = " << Name << " ";
|
||||
else
|
||||
Inst << "\tReg" << GetPhys(Node) << " = " << Name << " ";
|
||||
}
|
||||
else {
|
||||
Inst << "\t" << Name << " ";
|
||||
}
|
||||
|
||||
uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
uint32_t ArgNode = IROp->Args[i].ID();
|
||||
uint64_t PhysReg = RAPass->GetNodeRegister(ArgNode);
|
||||
if (PhysReg >= GPRPairBase)
|
||||
Inst << "Pair" << GetPhys(ArgNode) << (i + 1 == NumArgs ? "" : ", ");
|
||||
else if (PhysReg >= XMMBase)
|
||||
Inst << "XMM" << GetPhys(ArgNode) << (i + 1 == NumArgs ? "" : ", ");
|
||||
else
|
||||
Inst << "Reg" << GetPhys(ArgNode) << (i + 1 == NumArgs ? "" : ", ");
|
||||
}
|
||||
|
||||
LogMan::Msg::D("%s", Inst.str().c_str());
|
||||
}
|
||||
#endif
|
||||
uint32_t ID = IR->GetID(CodeNode);
|
||||
|
||||
// Execute handler
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
(this->*Handler)(IROp, ID);
|
||||
}
|
||||
}
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
if (PendingTargetLabel)
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
{
|
||||
jmp(*PendingTargetLabel, T_NEAR);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
uint32_t Node = IR->GetID(BlockNode);
|
||||
auto IsTarget = JumpTargets.find(Node);
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
}
|
||||
|
||||
// if there is a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
|
||||
{
|
||||
jmp(*PendingTargetLabel, T_NEAR);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
L(IsTarget->second);
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
#ifdef DEBUG_RA
|
||||
if (IROp->Op != IR::OP_BEGINBLOCK &&
|
||||
IROp->Op != IR::OP_CONDJUMP &&
|
||||
IROp->Op != IR::OP_JUMP) {
|
||||
std::stringstream Inst;
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
|
||||
if (IROp->HasDest) {
|
||||
uint64_t PhysReg = RAPass->GetNodeRegister(Node);
|
||||
if (PhysReg >= GPRPairBase)
|
||||
Inst << "\tPair" << GetPhys(Node) << " = " << Name << " ";
|
||||
else if (PhysReg >= XMMBase)
|
||||
Inst << "\tXMM" << GetPhys(Node) << " = " << Name << " ";
|
||||
else
|
||||
Inst << "\tReg" << GetPhys(Node) << " = " << Name << " ";
|
||||
}
|
||||
else {
|
||||
Inst << "\t" << Name << " ";
|
||||
}
|
||||
|
||||
uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
uint32_t ArgNode = IROp->Args[i].ID();
|
||||
uint64_t PhysReg = RAPass->GetNodeRegister(ArgNode);
|
||||
if (PhysReg >= GPRPairBase)
|
||||
Inst << "Pair" << GetPhys(ArgNode) << (i + 1 == NumArgs ? "" : ", ");
|
||||
else if (PhysReg >= XMMBase)
|
||||
Inst << "XMM" << GetPhys(ArgNode) << (i + 1 == NumArgs ? "" : ", ");
|
||||
else
|
||||
Inst << "Reg" << GetPhys(ArgNode) << (i + 1 == NumArgs ? "" : ", ");
|
||||
}
|
||||
|
||||
LogMan::Msg::D("%s", Inst.str().c_str());
|
||||
}
|
||||
#endif
|
||||
uint32_t ID = IR->GetID(CodeNode);
|
||||
|
||||
// Execute handler
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
(this->*Handler)(IROp, ID);
|
||||
}
|
||||
}
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
if (PendingTargetLabel)
|
||||
{
|
||||
jmp(*PendingTargetLabel, T_NEAR);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
void *Exit = getCurr<void*>();
|
||||
this->IR = nullptr;
|
||||
|
||||
@@ -915,7 +1144,7 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
je(NoBlock);
|
||||
|
||||
mov (rax, rdx);
|
||||
and(rax, 0x0FFF);
|
||||
and_(rax, 0x0FFF);
|
||||
|
||||
shl(rax, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
|
||||
@@ -1002,17 +1231,6 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Interpreter fallback helper code
|
||||
ThreadSharedData.InterpreterFallbackHelperAddress = getCurr<void*>();
|
||||
// This will get called so our stack is now misaligned
|
||||
mov(rax, reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR));
|
||||
mov(rdi, STATE);
|
||||
call(rax);
|
||||
|
||||
jmp(LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
// Signal return handler
|
||||
ThreadSharedData.SignalHandlerReturnAddress = getCurr<uint64_t>();
|
||||
|
||||
@@ -51,15 +51,15 @@ namespace FEXCore::CPU {
|
||||
using namespace Xbyak::util;
|
||||
const std::array<Xbyak::Reg, 9> RA64 = { rsi, r8, r9, r10, r11, rbp, r12, r13, r15 };
|
||||
const std::array<std::pair<Xbyak::Reg, Xbyak::Reg>, 4> RA64Pair = {{ {rsi, r8}, {r9, r10}, {r11, rbp}, {r12, r13} }};
|
||||
const std::array<Xbyak::Reg, 11> RAXMM = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 };
|
||||
const std::array<Xbyak::Xmm, 11> RAXMM_x = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 };
|
||||
const std::array<Xbyak::Reg, 11> RAXMM = { xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10, xmm11};
|
||||
const std::array<Xbyak::Xmm, 11> RAXMM_x = { xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10, xmm11};
|
||||
|
||||
class JITCore final : public CPUBackend, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
explicit JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer, bool CompileThread);
|
||||
~JITCore() override;
|
||||
std::string GetName() override { return "JIT"; }
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -79,7 +79,7 @@ private:
|
||||
Label* PendingTargetLabel{};
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
FEXCore::IR::IRListView<true> const *IR;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
|
||||
std::unordered_map<IR::OrderedNodeWrapper::NodeOffsetType, Label> JumpTargets;
|
||||
Xbyak::util::Cpu Features{};
|
||||
@@ -126,6 +126,7 @@ private:
|
||||
Xbyak::RegExp GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr);
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value);
|
||||
|
||||
void CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread);
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
@@ -167,8 +168,6 @@ private:
|
||||
uint32_t SignalHandlerRefCounter{};
|
||||
|
||||
struct CompilerSharedData {
|
||||
void *InterpreterFallbackHelperAddress;
|
||||
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
@@ -198,6 +197,9 @@ private:
|
||||
void RegisterMoveHandlers();
|
||||
void RegisterVectorHandlers();
|
||||
void RegisterEncryptionHandlers();
|
||||
|
||||
void PushRegs();
|
||||
void PopRegs();
|
||||
#define DEF_OP(x) void Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
|
||||
///< Unhandled handler
|
||||
@@ -208,8 +210,10 @@ private:
|
||||
|
||||
///< ALU Ops
|
||||
DEF_OP(TruncElementPair);
|
||||
DEF_OP(Constant);
|
||||
DEF_OP(Constant);
|
||||
DEF_OP(EntrypointOffset);
|
||||
DEF_OP(InlineConstant);
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
|
||||
@@ -48,7 +48,7 @@ DEF_OP(Break) {
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
|
||||
// This jump target needs to be a constant offset here
|
||||
mov(TMP1, ThreadPauseHandlerAddress);
|
||||
jmp(TMP1);
|
||||
@@ -89,10 +89,10 @@ DEF_OP(SetRoundingMode) {
|
||||
mov(TMP1.cvt32(), dword [rsp]);
|
||||
|
||||
// Insert the new rounding mode
|
||||
and(TMP1.cvt32(), ~(0b111 << 13));
|
||||
and_(TMP1.cvt32(), ~(0b111 << 13));
|
||||
mov(TMP2.cvt32(), Src);
|
||||
shl(TMP2.cvt32(), 13);
|
||||
or(TMP1.cvt32(), TMP2.cvt32());
|
||||
or_(TMP1.cvt32(), TMP2.cvt32());
|
||||
|
||||
// Store it to mxcsr
|
||||
// Only loads from memory
|
||||
|
||||
@@ -28,18 +28,21 @@ DEF_OP(CreateElementPair) {
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> Dst;
|
||||
Xbyak::Reg RegFirst;
|
||||
Xbyak::Reg RegSecond;
|
||||
Xbyak::Reg RegTmp;
|
||||
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetSrc<RA_32>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetSrc<RA_32>(Op->Header.Args[1].ID());
|
||||
RegTmp = eax;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetSrc<RA_64>(Op->Header.Args[1].ID());
|
||||
RegTmp = rax;
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::A("Unknown Size"); break;
|
||||
@@ -52,7 +55,9 @@ DEF_OP(CreateElementPair) {
|
||||
mov(Dst.second, RegSecond);
|
||||
mov(Dst.first, RegFirst);
|
||||
} else {
|
||||
LogMan::Msg::A("Unhandled CreateElementPair");
|
||||
mov(RegTmp, RegFirst);
|
||||
mov(Dst.second, RegSecond);
|
||||
mov(Dst.first, RegTmp);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+16
-13
@@ -262,17 +262,17 @@ DEF_OP(VAddP) {
|
||||
vpaddw(GetDst(Node), xmm15, xmm14);
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
vpunpcklbw(xmm11, xmm15, xmm14);
|
||||
vpunpcklbw(xmm0, xmm15, xmm14);
|
||||
vpunpckhbw(xmm12, xmm15, xmm14);
|
||||
|
||||
vpunpcklbw(xmm15, xmm11, xmm12);
|
||||
vpunpckhbw(xmm14, xmm11, xmm12);
|
||||
vpunpcklbw(xmm15, xmm0, xmm12);
|
||||
vpunpckhbw(xmm14, xmm0, xmm12);
|
||||
|
||||
vpunpcklbw(xmm11, xmm15, xmm14);
|
||||
vpunpcklbw(xmm0, xmm15, xmm14);
|
||||
vpunpckhbw(xmm12, xmm15, xmm14);
|
||||
|
||||
vpunpcklbw(xmm15, xmm11, xmm12);
|
||||
vpunpckhbw(xmm14, xmm11, xmm12);
|
||||
vpunpcklbw(xmm15, xmm0, xmm12);
|
||||
vpunpckhbw(xmm14, xmm0, xmm12);
|
||||
|
||||
vpaddb(GetDst(Node), xmm15, xmm14);
|
||||
break;
|
||||
@@ -291,17 +291,17 @@ DEF_OP(VAddP) {
|
||||
movdqu(xmm15, GetSrc(Op->Header.Args[0].ID()));
|
||||
movdqu(xmm14, GetSrc(Op->Header.Args[1].ID()));
|
||||
|
||||
vpunpcklbw(xmm11, xmm15, xmm14);
|
||||
vpunpcklbw(xmm0, xmm15, xmm14);
|
||||
vpunpckhbw(xmm12, xmm15, xmm14);
|
||||
|
||||
vpunpcklbw(xmm15, xmm11, xmm12);
|
||||
vpunpckhbw(xmm14, xmm11, xmm12);
|
||||
vpunpcklbw(xmm15, xmm0, xmm12);
|
||||
vpunpckhbw(xmm14, xmm0, xmm12);
|
||||
|
||||
vpunpcklbw(xmm11, xmm15, xmm14);
|
||||
vpunpcklbw(xmm0, xmm15, xmm14);
|
||||
vpunpckhbw(xmm12, xmm15, xmm14);
|
||||
|
||||
vpunpcklbw(xmm15, xmm11, xmm12);
|
||||
vpunpckhbw(xmm14, xmm11, xmm12);
|
||||
vpunpcklbw(xmm15, xmm0, xmm12);
|
||||
vpunpckhbw(xmm14, xmm0, xmm12);
|
||||
|
||||
vpaddb(GetDst(Node), xmm15, xmm14);
|
||||
break;
|
||||
@@ -912,7 +912,10 @@ DEF_OP(VZip2) {
|
||||
}
|
||||
|
||||
DEF_OP(VBSL) {
|
||||
LogMan::Msg::A("Unimplemented");
|
||||
auto Op = IROp->C<IR::IROp_VBSL>();
|
||||
vpand(xmm0, GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vpandn(xmm12, GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[2].ID()));
|
||||
vpor(GetDst(Node), xmm0, xmm12);
|
||||
}
|
||||
|
||||
DEF_OP(VCMPEQ) {
|
||||
|
||||
+8
-1
@@ -35,10 +35,16 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode) {
|
||||
std::map<uint64_t, std::vector<uint64_t>> CodePages;
|
||||
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode, uint64_t Start, uint64_t Length) {
|
||||
auto InsertPoint = BlockList.emplace(Address, (uintptr_t)HostCode);
|
||||
LogMan::Throw::A(InsertPoint.second == true, "Dupplicate block mapping added");
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
CodePages[CurrentPage].push_back(Address);
|
||||
}
|
||||
|
||||
// no need to update L1 or L2, they will get updated on first lookup
|
||||
}
|
||||
|
||||
@@ -197,6 +203,7 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
std::map<BlockLinkTag, std::function<void()>> BlockLinks;
|
||||
std::map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
|
||||
+193
-131
@@ -78,7 +78,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
LogMan::Msg::D("Unhandled OSABI syscall");
|
||||
}
|
||||
|
||||
auto NewRIP = _Constant(GPRSize * 8, Op->PC);
|
||||
auto NewRIP = GetDynamicPC(Op, -Op->InstSize);
|
||||
_StoreContext(GPRClass, GPRSize, offsetof(FEXCore::Core::CPUState, rip), NewRIP);
|
||||
|
||||
auto SyscallOp = _Syscall(
|
||||
@@ -95,14 +95,13 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
|
||||
const char *name;
|
||||
uint8_t *sha256;
|
||||
|
||||
name = (const char*)(Op->PC + 2);
|
||||
sha256 = (uint8_t *)(Op->PC + 2);
|
||||
|
||||
_Thunk(
|
||||
_LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RDI]), GPRClass),
|
||||
name,
|
||||
(uintptr_t)CTX->ThunkHandler->LookupThunk(name)
|
||||
*reinterpret_cast<SHA256Sum*>(sha256)
|
||||
);
|
||||
|
||||
auto Constant = _Constant(GPRSize);
|
||||
@@ -318,6 +317,7 @@ void OpDispatchBuilder::SecondaryALUOp(OpcodeArgs) {
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
void OpDispatchBuilder::ADCOp(OpcodeArgs) {
|
||||
LogMan::Throw::A(!DestIsLockedMem(Op), "Can't handle LOCK on ADC\n");
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
|
||||
@@ -337,6 +337,7 @@ void OpDispatchBuilder::ADCOp(OpcodeArgs) {
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
void OpDispatchBuilder::SBBOp(OpcodeArgs) {
|
||||
LogMan::Throw::A(!DestIsLockedMem(Op), "Can't handle LOCK on SBB\n");
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
|
||||
@@ -637,12 +638,12 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) {
|
||||
_InvalidateFlags(~0UL); // all flags
|
||||
}
|
||||
|
||||
auto ConstantPC = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
auto ConstantPC = GetDynamicPC(Op);
|
||||
|
||||
OrderedNode *JMPPCOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
OrderedNode *NewRIP = _Add(JMPPCOffset, ConstantPC);
|
||||
auto ConstantPCReturn = _Constant(Op->PC + Op->InstSize);
|
||||
OrderedNode *NewRIP = _Add(ConstantPC, JMPPCOffset);
|
||||
auto ConstantPCReturn = GetDynamicPC(Op);
|
||||
|
||||
auto ConstantSize = _Constant(GPRSize);
|
||||
auto OldSP = _LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RSP]), GPRClass);
|
||||
@@ -665,7 +666,7 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) {
|
||||
uint8_t Size = GetSrcSize(Op);
|
||||
OrderedNode *JMPPCOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
auto ConstantPCReturn = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
auto ConstantPCReturn = GetDynamicPC(Op);
|
||||
|
||||
auto ConstantSize = _Constant(Size);
|
||||
auto OldSP = _LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RSP]), GPRClass);
|
||||
@@ -961,10 +962,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
SetTrueJumpTarget(CondJump, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
|
||||
auto RIPOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto RIPTargetConst = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
|
||||
auto NewRIP = _Add(RIPOffset, RIPTargetConst);
|
||||
auto NewRIP = GetDynamicPC(Op, Op->Src[0].TypeLiteral.Literal);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
@@ -982,7 +980,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
|
||||
// Leave block
|
||||
auto RIPTargetConst = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
auto RIPTargetConst = GetDynamicPC(Op);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPTargetConst);
|
||||
@@ -1025,7 +1023,7 @@ void OpDispatchBuilder::CondJUMPRCXOp(OpcodeArgs) {
|
||||
SetTrueJumpTarget(CondJump, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
|
||||
auto NewRIP = _Constant(GPRSize * 8, Target);
|
||||
auto NewRIP = GetDynamicPC(Op, Op->Src[0].TypeLiteral.Literal);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
@@ -1043,7 +1041,7 @@ void OpDispatchBuilder::CondJUMPRCXOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
|
||||
// Leave block
|
||||
auto RIPTargetConst = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
auto RIPTargetConst = GetDynamicPC(Op);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPTargetConst);
|
||||
@@ -1102,7 +1100,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
SetTrueJumpTarget(CondJump, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
|
||||
auto NewRIP = _Constant(GPRSize * 8, Target);
|
||||
auto NewRIP = GetDynamicPC(Op, Op->Src[1].TypeLiteral.Literal);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
@@ -1120,7 +1118,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
|
||||
// Leave block
|
||||
auto RIPTargetConst = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
auto RIPTargetConst = GetDynamicPC(Op);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPTargetConst);
|
||||
@@ -1148,7 +1146,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
auto JumpTarget = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetJumpTarget(Jump, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
_ExitFunction(_Constant(GPRSize * 8, Target));
|
||||
_ExitFunction(GetDynamicPC(Op, Op->Src[0].TypeLiteral.Literal));
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -1158,7 +1156,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
// This source is a literal
|
||||
auto RIPOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
auto RIPTargetConst = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
auto RIPTargetConst = GetDynamicPC(Op);
|
||||
auto NewRIP = _Add(RIPOffset, RIPTargetConst);
|
||||
|
||||
// Store the new RIP
|
||||
@@ -1871,6 +1869,10 @@ void OpDispatchBuilder::ASHRImmediateOp(OpcodeArgs) {
|
||||
else
|
||||
Shift &= 0x1F;
|
||||
|
||||
if (Size < 32) {
|
||||
Dest = _Sbfe(Size, 0, Dest);
|
||||
}
|
||||
|
||||
OrderedNode *Src = _Constant(Size, Shift);
|
||||
OrderedNode *Result = _Ashr(Dest, Src);
|
||||
|
||||
@@ -2483,14 +2485,24 @@ void OpDispatchBuilder::BTROp(OpcodeArgs) {
|
||||
|
||||
// Now add the addresses together and load the memory
|
||||
OrderedNode *MemoryLocation = _Add(Dest, Src);
|
||||
OrderedNode *Value = _LoadMemAutoTSO(GPRClass, 1, MemoryLocation, 1);
|
||||
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Value, BitSelect);
|
||||
OrderedNode *BitMask = _Lshl(_Constant(1), BitSelect);
|
||||
BitMask = _Not(BitMask);
|
||||
Value = _And(Value, BitMask);
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
// XXX: Technically this can optimize to an AArch64 ldclralb
|
||||
// We don't current support this IR op though
|
||||
Result = _AtomicFetchAnd(MemoryLocation, BitMask, 1);
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Result, BitSelect);
|
||||
}
|
||||
else {
|
||||
OrderedNode *Value = _LoadMemAutoTSO(GPRClass, 1, MemoryLocation, 1);
|
||||
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Value, BitSelect);
|
||||
Value = _And(Value, BitMask);
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
}
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
}
|
||||
@@ -2550,13 +2562,21 @@ void OpDispatchBuilder::BTSOp(OpcodeArgs) {
|
||||
|
||||
// Now add the addresses together and load the memory
|
||||
OrderedNode *MemoryLocation = _Add(Dest, Src);
|
||||
OrderedNode *Value = _LoadMemAutoTSO(GPRClass, 1, MemoryLocation, 1);
|
||||
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Value, BitSelect);
|
||||
OrderedNode *BitMask = _Lshl(_Constant(1), BitSelect);
|
||||
Value = _Or(Value, BitMask);
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
Result = _AtomicFetchOr(MemoryLocation, BitMask, 1);
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Result, BitSelect);
|
||||
}
|
||||
else {
|
||||
OrderedNode *Value = _LoadMemAutoTSO(GPRClass, 1, MemoryLocation, 1);
|
||||
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Value, BitSelect);
|
||||
Value = _Or(Value, BitMask);
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
}
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
}
|
||||
@@ -2615,13 +2635,21 @@ void OpDispatchBuilder::BTCOp(OpcodeArgs) {
|
||||
|
||||
// Now add the addresses together and load the memory
|
||||
OrderedNode *MemoryLocation = _Add(Dest, Src);
|
||||
OrderedNode *Value = _LoadMemAutoTSO(GPRClass, 1, MemoryLocation, 1);
|
||||
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Value, BitSelect);
|
||||
OrderedNode *BitMask = _Lshl(_Constant(1), BitSelect);
|
||||
Value = _Xor(Value, BitMask);
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
Result = _AtomicFetchXor(MemoryLocation, BitMask, 1);
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Result, BitSelect);
|
||||
}
|
||||
else {
|
||||
OrderedNode *Value = _LoadMemAutoTSO(GPRClass, 1, MemoryLocation, 1);
|
||||
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Value, BitSelect);
|
||||
Value = _Xor(Value, BitMask);
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
}
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
}
|
||||
@@ -2749,6 +2777,7 @@ void OpDispatchBuilder::MULOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::NOTOp(OpcodeArgs) {
|
||||
LogMan::Throw::A(!DestIsLockedMem(Op), "Can't handle LOCK on NOT\n");
|
||||
uint8_t Size = GetSrcSize(Op);
|
||||
OrderedNode *MaskConst{};
|
||||
if (Size == 8) {
|
||||
@@ -3528,6 +3557,7 @@ void OpDispatchBuilder::POPFOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::NEGOp(OpcodeArgs) {
|
||||
LogMan::Throw::A(!DestIsLockedMem(Op), "Can't handle LOCK on NEG\n");
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
auto ZeroConst = _Constant(0);
|
||||
OrderedNode *Result = _Sub(ZeroConst, Dest);
|
||||
@@ -4235,7 +4265,6 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
|
||||
// This is our source register
|
||||
OrderedNode *Src2 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src3 = _LoadContext(Size, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), GPRClass);
|
||||
// 0x80014000
|
||||
// 0x80064000
|
||||
// 0x80064000
|
||||
@@ -4245,22 +4274,29 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
if (Op->Dest.TypeNone.Type == FEXCore::X86Tables::DecodedOperand::TYPE_GPR) {
|
||||
OrderedNode *Src1{};
|
||||
OrderedNode *Src1Lower{};
|
||||
|
||||
OrderedNode *Src3{};
|
||||
OrderedNode *Src3Lower{};
|
||||
if (GPRSize == 8 && Size == 4) {
|
||||
Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GPRSize, Op->Flags, -1);
|
||||
Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags, -1);
|
||||
Src1Lower = _Bfe(4, 32, 0, Src1);
|
||||
Src3 = _LoadContext(8, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), GPRClass);
|
||||
Src3Lower = _Bfe(4, 32, 0, Src3);
|
||||
}
|
||||
else {
|
||||
Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, Size, Op->Flags, -1);
|
||||
Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, Size, Op->Flags, -1);
|
||||
Src1Lower = Src1;
|
||||
Src3 = _LoadContext(Size, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), GPRClass);
|
||||
Src3Lower = Src3;
|
||||
}
|
||||
|
||||
// If our destination is a GPR then this behaves differently
|
||||
// RAX = RAX == Op1 ? RAX : Op1
|
||||
// AKA if they match then don't touch RAX value
|
||||
// Otherwise set it to the rm operand
|
||||
OrderedNode *RAXResult = _Select(FEXCore::IR::COND_EQ,
|
||||
OrderedNode *CASResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Src1Lower, Src3,
|
||||
Src3, Src1Lower);
|
||||
Src3Lower, Src1Lower);
|
||||
|
||||
// Op1 = RAX == Op1 ? Op2 : Op1
|
||||
// If they match then set the rm operand to the input
|
||||
@@ -4269,28 +4305,42 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
Src1Lower, Src3,
|
||||
Src2, Src1);
|
||||
|
||||
// ZF = RAX == Op1 ? 1 : 0
|
||||
// Result of compare
|
||||
OrderedNode *ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Src1, Src3,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
// Set ZF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFResult);
|
||||
|
||||
// Store in to GPR Dest
|
||||
// Have to make sure this is after the result store in RAX for when Dest == RAX
|
||||
if (GPRSize == 8 && Size == 4) {
|
||||
// This allows us to only hit the ZEXT case on failure
|
||||
OrderedNode *RAXResult = _Select(FEXCore::IR::COND_EQ,
|
||||
CASResult, Src3Lower,
|
||||
Src3, Src3Lower);
|
||||
|
||||
// When the size is 4 we need to make sure not zext the GPR when the comparison fails
|
||||
_StoreContext(GPRClass, GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), RAXResult);
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, DestResult, GPRSize, -1);
|
||||
}
|
||||
else {
|
||||
_StoreContext(GPRClass, Size, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), RAXResult);
|
||||
_StoreContext(GPRClass, Size, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), CASResult);
|
||||
StoreResult(GPRClass, Op, DestResult, -1);
|
||||
}
|
||||
|
||||
auto Size = GetDstSize(Op) * 8;
|
||||
|
||||
OrderedNode *Result = _Sub(Src3Lower, CASResult);
|
||||
if (Size < 32)
|
||||
Result = _Bfe(Size, 0, Result);
|
||||
|
||||
GenerateFlags_SUB(Op, Result, Src3Lower, CASResult);
|
||||
}
|
||||
else {
|
||||
OrderedNode *Src3{};
|
||||
OrderedNode *Src3Lower{};
|
||||
if (GPRSize == 8 && Size == 4) {
|
||||
Src3 = _LoadContext(8, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), GPRClass);
|
||||
Src3Lower = _Bfe(4, 32, 0, Src3);
|
||||
}
|
||||
else {
|
||||
Src3 = _LoadContext(Size, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), GPRClass);
|
||||
Src3Lower = Src3;
|
||||
}
|
||||
// If this is a memory location then we want the pointer to it
|
||||
OrderedNode *Src1 = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
|
||||
@@ -4300,26 +4350,27 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
// Third operand must be a calculated guest memory address
|
||||
OrderedNode *CASResult = _CAS(Src3, Src2, Src1);
|
||||
OrderedNode *CASResult = _CAS(Src3Lower, Src2, Src1);
|
||||
OrderedNode *RAXResult = CASResult;
|
||||
|
||||
// If our CASResult(OldMem value) is equal to our comparison
|
||||
// Then we managed to set the memory
|
||||
OrderedNode *ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
CASResult, Src3,
|
||||
OneConst, ZeroConst);
|
||||
if (GPRSize == 8 && Size == 4) {
|
||||
// This allows us to only hit the ZEXT case on failure
|
||||
RAXResult = _Select(FEXCore::IR::COND_EQ,
|
||||
CASResult, Src3Lower,
|
||||
Src3, CASResult);
|
||||
Size = 8;
|
||||
}
|
||||
|
||||
// RAX gets the result of the CAS op
|
||||
_StoreContext(GPRClass, Size, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), CASResult);
|
||||
_StoreContext(GPRClass, Size, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]), RAXResult);
|
||||
|
||||
auto Size = GetDstSize(Op) * 8;
|
||||
OrderedNode *Result = _Sub(CASResult, Src3);
|
||||
|
||||
OrderedNode *Result = _Sub(Src3Lower, CASResult);
|
||||
if (Size < 32)
|
||||
Result = _Bfe(Size, 0, Result);
|
||||
|
||||
GenerateFlags_SUB(Op, Result, CASResult, Src3);
|
||||
|
||||
// Set ZF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFResult);
|
||||
GenerateFlags_SUB(Op, Result, Src3Lower, CASResult);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4404,8 +4455,9 @@ void OpDispatchBuilder::CreateJumpBlocks(std::vector<FEXCore::Frontend::Decoder:
|
||||
|
||||
void OpDispatchBuilder::BeginFunction(uint64_t RIP, std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks) {
|
||||
Entry = RIP;
|
||||
auto IRHeader = _IRHeader(InvalidNode, RIP, 0, false);
|
||||
auto IRHeader = _IRHeader(InvalidNode, RIP, 0);
|
||||
Current_Header = IRHeader.first;
|
||||
Current_HeaderNode = IRHeader;
|
||||
CreateJumpBlocks(Blocks);
|
||||
|
||||
auto Block = GetNewJumpBlock(RIP);
|
||||
@@ -4571,7 +4623,7 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
}
|
||||
else if (Operand.TypeNone.Type == FEXCore::X86Tables::DecodedOperand::TYPE_RIP_RELATIVE) {
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
Src = _Constant(GPRSize * 8, Operand.TypeRIPLiteral.Literal.s + Op->PC + Op->InstSize);
|
||||
Src = GetDynamicPC(Op, Operand.TypeRIPLiteral.Literal.s);
|
||||
}
|
||||
else {
|
||||
// 32bit this isn't RIP relative but instead absolute
|
||||
@@ -4646,6 +4698,11 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
return Src;
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::GetDynamicPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset) {
|
||||
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
|
||||
return _EntrypointOffset(Op->PC + Op->InstSize + Offset - Current_Header->Entry, GPRSize);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad) {
|
||||
uint8_t OpSize = GetSrcSize(Op);
|
||||
return LoadSource_WithOpSize(Class, Op, Operand, OpSize, Flags, Align, LoadData, ForceLoad);
|
||||
@@ -4709,7 +4766,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
}
|
||||
else if (Operand.TypeNone.Type == FEXCore::X86Tables::DecodedOperand::TYPE_RIP_RELATIVE) {
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
MemStoreDst = _Constant(GPRSize * 8, Operand.TypeRIPLiteral.Literal.s + Op->PC + Op->InstSize);
|
||||
MemStoreDst = GetDynamicPC(Op, Operand.TypeRIPLiteral.Literal.s);
|
||||
}
|
||||
else {
|
||||
// 32bit this isn't RIP relative but instead absolute
|
||||
@@ -5799,7 +5856,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
BlockSetRIP = setRIP;
|
||||
|
||||
// We want to set RIP to the next instruction after HLT/INT3
|
||||
auto NewRIP = _Constant(Op->PC + Op->InstSize);
|
||||
auto NewRIP = GetDynamicPC(Op);
|
||||
_StoreContext(GPRClass, GPRSize, offsetof(FEXCore::Core::CPUState, rip), NewRIP);
|
||||
}
|
||||
|
||||
@@ -6298,7 +6355,6 @@ void OpDispatchBuilder::SetX87Top(OrderedNode *Value) {
|
||||
|
||||
template<size_t width>
|
||||
void OpDispatchBuilder::FLD(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
// Update TOP
|
||||
auto orig_top = GetX87Top();
|
||||
@@ -6333,7 +6389,6 @@ void OpDispatchBuilder::FLD(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
// Update TOP
|
||||
auto orig_top = GetX87Top();
|
||||
@@ -6348,7 +6403,6 @@ void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
auto data = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
@@ -6363,7 +6417,6 @@ void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
|
||||
template<uint64_t Lower, uint32_t Upper>
|
||||
void OpDispatchBuilder::FLD_Const(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
// Update TOP
|
||||
auto orig_top = GetX87Top();
|
||||
auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7));
|
||||
@@ -6378,7 +6431,6 @@ void OpDispatchBuilder::FLD_Const(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
// Update TOP
|
||||
auto orig_top = GetX87Top();
|
||||
@@ -6418,7 +6470,6 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
|
||||
template<size_t width>
|
||||
void OpDispatchBuilder::FST(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
auto data = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
@@ -6436,14 +6487,14 @@ void OpDispatchBuilder::FST(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template<bool Truncate>
|
||||
void OpDispatchBuilder::FIST(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
OrderedNode *data = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
data = _F80CVTInt(data, Size);
|
||||
data = _F80CVTInt(data, Truncate, Size);
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, 1);
|
||||
|
||||
@@ -6455,7 +6506,6 @@ void OpDispatchBuilder::FIST(OpcodeArgs) {
|
||||
|
||||
template <size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FADD(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode *StackLocation = top;
|
||||
@@ -6467,12 +6517,13 @@ void OpDispatchBuilder::FADD(OpcodeArgs) {
|
||||
|
||||
if (Op->Src[0].TypeNone.Type != 0) {
|
||||
// Memory arg
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
if (width == 16 || width == 32 || width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -6500,7 +6551,6 @@ void OpDispatchBuilder::FADD(OpcodeArgs) {
|
||||
|
||||
template<size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FMUL(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode *StackLocation = top;
|
||||
@@ -6511,13 +6561,14 @@ void OpDispatchBuilder::FMUL(OpcodeArgs) {
|
||||
|
||||
if (Op->Src[0].TypeNone.Type != 0) {
|
||||
// Memory arg
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
if (width == 16 || width == 32 || width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -6547,7 +6598,6 @@ void OpDispatchBuilder::FMUL(OpcodeArgs) {
|
||||
|
||||
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FDIV(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode *StackLocation = top;
|
||||
@@ -6558,13 +6608,14 @@ void OpDispatchBuilder::FDIV(OpcodeArgs) {
|
||||
|
||||
if (Op->Src[0].TypeNone.Type != 0) {
|
||||
// Memory arg
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
if (width == 16 || width == 32 || width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -6600,7 +6651,6 @@ void OpDispatchBuilder::FDIV(OpcodeArgs) {
|
||||
|
||||
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FSUB(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode *StackLocation = top;
|
||||
@@ -6611,13 +6661,14 @@ void OpDispatchBuilder::FSUB(OpcodeArgs) {
|
||||
|
||||
if (Op->Src[0].TypeNone.Type != 0) {
|
||||
// Memory arg
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
if (width == 16 || width == 32 || width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -6651,7 +6702,6 @@ void OpDispatchBuilder::FSUB(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FCHS(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
@@ -6668,7 +6718,6 @@ void OpDispatchBuilder::FCHS(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FABS(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
@@ -6685,7 +6734,6 @@ void OpDispatchBuilder::FABS(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FTST(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
@@ -6711,7 +6759,6 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FRNDINT(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
@@ -6723,7 +6770,6 @@ void OpDispatchBuilder::FRNDINT(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FXTRACT(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7));
|
||||
@@ -6740,14 +6786,16 @@ void OpDispatchBuilder::FXTRACT(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto NewFCW = _Constant(16, 0x037);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
|
||||
|
||||
SetX87Top(_Constant(0));
|
||||
}
|
||||
|
||||
template<size_t width, bool Integer, OpDispatchBuilder::FCOMIFlags whichflags, bool poptwice>
|
||||
void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
auto mask = _Constant(7);
|
||||
@@ -6757,12 +6805,13 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
|
||||
if (Op->Src[0].TypeNone.Type != 0) {
|
||||
// Memory arg
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
if (width == 16 || width == 32 || width == 64) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -6810,7 +6859,6 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* arg;
|
||||
@@ -6830,7 +6878,6 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* arg;
|
||||
@@ -6854,7 +6901,6 @@ void OpDispatchBuilder::FST(OpcodeArgs) {
|
||||
|
||||
template<FEXCore::IR::IROps IROp>
|
||||
void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
@@ -6869,7 +6915,6 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
|
||||
|
||||
template<FEXCore::IR::IROps IROp>
|
||||
void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
auto top = GetX87Top();
|
||||
|
||||
auto mask = _Constant(7);
|
||||
@@ -6882,6 +6927,11 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if (IROp == IR::OP_F80FPREM) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
}
|
||||
@@ -6900,7 +6950,6 @@ void OpDispatchBuilder::X87ModifySTP(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87SinCos(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7));
|
||||
@@ -6918,7 +6967,6 @@ void OpDispatchBuilder::X87SinCos(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FYL2X(OpcodeArgs) {
|
||||
bool Plus1 = Op->OP == 0x01F9; // FYL2XP
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7));
|
||||
@@ -6942,7 +6990,6 @@ void OpDispatchBuilder::X87FYL2X(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87TAN(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7));
|
||||
@@ -6963,7 +7010,6 @@ void OpDispatchBuilder::X87TAN(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87ATAN(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7));
|
||||
@@ -6979,10 +7025,13 @@ void OpDispatchBuilder::X87ATAN(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
|
||||
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
|
||||
@@ -7002,7 +7051,6 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
// 14 bytes for 16bit
|
||||
// 2 Bytes : FCW
|
||||
// 2 Bytes : FSW
|
||||
@@ -7026,8 +7074,8 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
|
||||
{
|
||||
// FCW store default
|
||||
_StoreMem(GPRClass, Size, Mem, _Constant(0x37F), Size);
|
||||
auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass);
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -7082,12 +7130,19 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, _Constant(0x37F), -1);
|
||||
auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass);
|
||||
|
||||
StoreResult(GPRClass, Op, FCW, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87LDSW(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
OrderedNode *NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
// Strip out the FSW information
|
||||
auto Top = _Bfe(3, 11, NewFSW);
|
||||
@@ -7105,7 +7160,6 @@ void OpDispatchBuilder::X87LDSW(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
// We must construct the FSW from our various bits
|
||||
OrderedNode *FSW = _Constant(0);
|
||||
auto Top = GetX87Top();
|
||||
@@ -7125,7 +7179,6 @@ void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
// 14 bytes for 16bit
|
||||
// 2 Bytes : FCW
|
||||
// 2 Bytes : FSW
|
||||
@@ -7150,8 +7203,8 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Top = GetX87Top();
|
||||
{
|
||||
// FCW store default
|
||||
_StoreMem(GPRClass, Size, Mem, _Constant(0x37F), Size);
|
||||
auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass);
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -7223,14 +7276,18 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, ST0Location, data, 1);
|
||||
ST0Location = _Add(ST0Location, _Constant(8));
|
||||
_VStoreMemElement(16, 2, ST0Location, data, 4, 1);
|
||||
auto topBytes = _VExtractElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, ST0Location, topBytes, 1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
|
||||
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
|
||||
@@ -7277,7 +7334,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Reg = _LoadMem(FPRClass, 8, ST0Location, 1);
|
||||
ST0Location = _Add(ST0Location, _Constant(8));
|
||||
Reg = _VLoadMemElement(16, 2, ST0Location, Reg, 4, 1);
|
||||
OrderedNode *RegHigh = _LoadMem(FPRClass, 2, ST0Location, 1);
|
||||
Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh);
|
||||
_StoreContextIndexed(Reg, Top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass);
|
||||
}
|
||||
|
||||
@@ -7300,7 +7358,6 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
Current_Header->ShouldInterpret = true;
|
||||
enum CompareType {
|
||||
COMPARE_ZERO,
|
||||
COMPARE_NOTZERO,
|
||||
@@ -7410,8 +7467,8 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
{
|
||||
// FCW store default
|
||||
_StoreMem(GPRClass, 2, Mem, _Constant(0x37F), 2);
|
||||
auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass);
|
||||
_StoreMem(GPRClass, 2, Mem, FCW, 2);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -7493,6 +7550,11 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
|
||||
|
||||
{
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
|
||||
auto NewFSW = _LoadMem(GPRClass, 2, MemLocation, 2);
|
||||
@@ -8163,7 +8225,7 @@ void OpDispatchBuilder::UnimplementedOp(OpcodeArgs) {
|
||||
|
||||
// We don't actually support this instruction
|
||||
// Multiblock may hit it though
|
||||
_StoreContext(GPRClass, GPRSize, offsetof(FEXCore::Core::CPUState, rip), _Constant(Op->PC));
|
||||
_StoreContext(GPRClass, GPRSize, offsetof(FEXCore::Core::CPUState, rip), GetDynamicPC(Op, -Op->InstSize));
|
||||
_Break(0, 0);
|
||||
BlockSetRIP = true;
|
||||
|
||||
@@ -8315,8 +8377,8 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x28, 2, &OpDispatchBuilder::MOVUPSOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true, false>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVAPSOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, true, false, true>},
|
||||
{0x2D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, true, false, false>},
|
||||
{0x2C, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, true, false, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, true, false, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<4>},
|
||||
{0x50, 1, &OpDispatchBuilder::MOVMSKOp<4>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4, false>},
|
||||
@@ -8605,8 +8667,8 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x28, 2, &OpDispatchBuilder::MOVAPSOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true, true>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVAPSOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, true, true>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, true, false>},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, true, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, true, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<8>},
|
||||
|
||||
{0x40, 16, &OpDispatchBuilder::CMOVOp},
|
||||
@@ -8835,7 +8897,7 @@ constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
{OPDReg(0xD9, 4) | 0x00, 8, &OpDispatchBuilder::X87LDENV},
|
||||
|
||||
{OPDReg(0xD9, 5) | 0x00, 8, &OpDispatchBuilder::NOPOp}, // XXX: stubbed FLDCW
|
||||
{OPDReg(0xD9, 5) | 0x00, 8, &OpDispatchBuilder::X87FLDCW}, // XXX: stubbed FLDCW
|
||||
|
||||
{OPDReg(0xD9, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSTENV},
|
||||
|
||||
@@ -8907,11 +8969,11 @@ constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
{OPDReg(0xDB, 0) | 0x00, 8, &OpDispatchBuilder::FILD},
|
||||
|
||||
{OPDReg(0xDB, 1) | 0x00, 8, &OpDispatchBuilder::FIST},
|
||||
{OPDReg(0xDB, 1) | 0x00, 8, &OpDispatchBuilder::FIST<true>},
|
||||
|
||||
{OPDReg(0xDB, 2) | 0x00, 8, &OpDispatchBuilder::FIST},
|
||||
{OPDReg(0xDB, 2) | 0x00, 8, &OpDispatchBuilder::FIST<false>},
|
||||
|
||||
{OPDReg(0xDB, 3) | 0x00, 8, &OpDispatchBuilder::FIST},
|
||||
{OPDReg(0xDB, 3) | 0x00, 8, &OpDispatchBuilder::FIST<false>},
|
||||
|
||||
// 4 = Invalid
|
||||
|
||||
@@ -8960,7 +9022,7 @@ constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
{OPDReg(0xDD, 0) | 0x00, 8, &OpDispatchBuilder::FLD<64>},
|
||||
|
||||
{OPDReg(0xDD, 1) | 0x00, 8, &OpDispatchBuilder::FIST},
|
||||
{OPDReg(0xDD, 1) | 0x00, 8, &OpDispatchBuilder::FIST<true>},
|
||||
|
||||
{OPDReg(0xDD, 2) | 0x00, 8, &OpDispatchBuilder::FST<64>},
|
||||
|
||||
@@ -9006,11 +9068,11 @@ constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
{OPDReg(0xDF, 0) | 0x00, 8, &OpDispatchBuilder::FILD},
|
||||
|
||||
{OPDReg(0xDF, 1) | 0x00, 8, &OpDispatchBuilder::FIST},
|
||||
{OPDReg(0xDF, 1) | 0x00, 8, &OpDispatchBuilder::FIST<true>},
|
||||
|
||||
{OPDReg(0xDF, 2) | 0x00, 8, &OpDispatchBuilder::FIST},
|
||||
{OPDReg(0xDF, 2) | 0x00, 8, &OpDispatchBuilder::FIST<false>},
|
||||
|
||||
{OPDReg(0xDF, 3) | 0x00, 8, &OpDispatchBuilder::FIST},
|
||||
{OPDReg(0xDF, 3) | 0x00, 8, &OpDispatchBuilder::FIST<false>},
|
||||
|
||||
{OPDReg(0xDF, 4) | 0x00, 8, &OpDispatchBuilder::FBLD},
|
||||
|
||||
@@ -9018,7 +9080,7 @@ constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
{OPDReg(0xDF, 6) | 0x00, 8, &OpDispatchBuilder::FBSTP},
|
||||
|
||||
{OPDReg(0xDF, 7) | 0x00, 8, &OpDispatchBuilder::FIST},
|
||||
{OPDReg(0xDF, 7) | 0x00, 8, &OpDispatchBuilder::FIST<false>},
|
||||
|
||||
// XXX: This should also set the x87 tag bits to empty
|
||||
// We don't support this currently, so just pop the stack
|
||||
|
||||
@@ -88,7 +88,8 @@ public:
|
||||
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
|
||||
// If we don't have a jump target to a new block then we have to leave
|
||||
// Set the RIP to the next instruction and leave
|
||||
_ExitFunction(_Constant(GPRSize * 8, NextRIP));
|
||||
auto RelocatedNextRIP = _EntrypointOffset(NextRIP - Current_Header->Entry, GPRSize);
|
||||
_ExitFunction(RelocatedNextRIP);
|
||||
}
|
||||
else if (it != JumpTargets.end()) {
|
||||
_Jump(it->second.BlockEntry);
|
||||
@@ -349,6 +350,8 @@ public:
|
||||
void FST(OpcodeArgs);
|
||||
|
||||
void FST(OpcodeArgs);
|
||||
|
||||
template<bool Truncate>
|
||||
void FIST(OpcodeArgs);
|
||||
|
||||
enum class OpResult {
|
||||
@@ -381,6 +384,7 @@ public:
|
||||
void X87TAN(OpcodeArgs);
|
||||
void X87ATAN(OpcodeArgs);
|
||||
void X87LDENV(OpcodeArgs);
|
||||
void X87FLDCW(OpcodeArgs);
|
||||
void X87FNSTENV(OpcodeArgs);
|
||||
void X87FSTCW(OpcodeArgs);
|
||||
void X87LDSW(OpcodeArgs);
|
||||
@@ -476,9 +480,11 @@ public:
|
||||
private:
|
||||
bool DecodeFailure{false};
|
||||
FEXCore::IR::IROp_IRHeader *Current_Header{};
|
||||
OrderedNode *Current_HeaderNode{};
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
|
||||
OrderedNode *GetDynamicPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false);
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align);
|
||||
|
||||
+12
-7
@@ -7,6 +7,7 @@
|
||||
|
||||
#include <string>
|
||||
#include <map>
|
||||
#include <array>
|
||||
#include <Interface/Context/Context.h>
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
@@ -23,13 +24,17 @@ static thread_local FEXCore::Core::InternalThreadState *Thread;
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
struct ExportEntry { const char* Name; ThunkedFunction* Fn; };
|
||||
struct ExportEntry { uint8_t *sha256; ThunkedFunction* Fn; };
|
||||
|
||||
class ThunkHandler_impl final: public ThunkHandler {
|
||||
std::shared_mutex ThunksMutex;
|
||||
|
||||
std::map<std::string, ThunkedFunction*> Thunks = {
|
||||
{ "fex:loadlib", &LoadLib}
|
||||
std::map<IR::SHA256Sum, ThunkedFunction*> Thunks = {
|
||||
{
|
||||
// sha256(fex:loadlib)
|
||||
{ 0x27, 0x7e, 0xb7, 0x69, 0x5b, 0xe9, 0xab, 0x12, 0x6e, 0xf7, 0x85, 0x9d, 0x4b, 0xc9, 0xa2, 0x44, 0x46, 0xcf, 0xbd, 0xb5, 0x87, 0x43, 0xef, 0x28, 0xa2, 0x65, 0xba, 0xfc, 0x89, 0x0f, 0x77, 0x80},
|
||||
&LoadLib
|
||||
}
|
||||
};
|
||||
|
||||
/*
|
||||
@@ -82,8 +87,8 @@ namespace FEXCore {
|
||||
std::unique_lock lk(That->ThunksMutex);
|
||||
|
||||
int i;
|
||||
for (i = 0; Exports[i].Name; i++) {
|
||||
That->Thunks[Exports[i].Name] = Exports[i].Fn;
|
||||
for (i = 0; Exports[i].sha256; i++) {
|
||||
That->Thunks[*reinterpret_cast<IR::SHA256Sum*>(Exports[i].sha256)] = Exports[i].Fn;
|
||||
}
|
||||
|
||||
LogMan::Msg::D("Loaded %d syms", i);
|
||||
@@ -92,11 +97,11 @@ namespace FEXCore {
|
||||
|
||||
public:
|
||||
|
||||
ThunkedFunction* LookupThunk(const char *Name) {
|
||||
ThunkedFunction* LookupThunk(const IR::SHA256Sum &sha256) {
|
||||
|
||||
std::shared_lock lk(ThunksMutex);
|
||||
|
||||
auto it = Thunks.find(Name);
|
||||
auto it = Thunks.find(sha256);
|
||||
|
||||
if (it != Thunks.end()) {
|
||||
return it->second;
|
||||
|
||||
+2
-1
@@ -1,4 +1,5 @@
|
||||
#pragma once
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
@@ -9,7 +10,7 @@ namespace FEXCore {
|
||||
|
||||
class ThunkHandler {
|
||||
public:
|
||||
virtual ThunkedFunction* LookupThunk(const char *name) = 0;
|
||||
virtual ThunkedFunction* LookupThunk(const IR::SHA256Sum &sha256) = 0;
|
||||
virtual void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
virtual ~ThunkHandler() { }
|
||||
|
||||
|
||||
+42
-9
@@ -80,8 +80,7 @@
|
||||
],
|
||||
"Args": [
|
||||
"uint64_t", "Entry",
|
||||
"uint32_t", "BlockCount",
|
||||
"bool", "ShouldInterpret"
|
||||
"uint32_t", "BlockCount"
|
||||
]
|
||||
},
|
||||
"CodeBlock": {
|
||||
@@ -133,17 +132,14 @@
|
||||
"Args": [
|
||||
"uint64_t", "CodeOriginalLow",
|
||||
"uint64_t", "CodeOriginalHigh",
|
||||
"uint64_t", "CodePtr",
|
||||
"int64_t", "Offset",
|
||||
"uint8_t", "CodeLength"
|
||||
]
|
||||
},
|
||||
|
||||
"RemoveCodeEntry": {
|
||||
"HasSideEffects": true,
|
||||
"OpClass": "Misc",
|
||||
"Args": [
|
||||
"uint64_t", "RIP"
|
||||
]
|
||||
"OpClass": "Misc"
|
||||
},
|
||||
|
||||
"GuestCallDirect": {
|
||||
@@ -284,6 +280,32 @@
|
||||
]
|
||||
},
|
||||
|
||||
"EntrypointOffset": {
|
||||
"Desc": ["Returns the <entrypoint> + Offset address"],
|
||||
"OpClass": "ALU",
|
||||
"HasDest": true,
|
||||
"DestClass": "GPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"Args": [
|
||||
"int64_t", "Offset"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize"
|
||||
]
|
||||
},
|
||||
|
||||
"InlineEntrypointOffset": {
|
||||
"Desc": ["Returns the <entrypoint> + Offset address"],
|
||||
"OpClass": "ALU",
|
||||
"DestSize": "RegisterSize",
|
||||
"Args": [
|
||||
"int64_t", "Offset"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize"
|
||||
]
|
||||
},
|
||||
|
||||
"Constant": {
|
||||
"Desc": ["Generates a 64bit constant inside of a GPR",
|
||||
"Unsupported to create a constant in FPR"
|
||||
@@ -644,8 +666,7 @@
|
||||
"ArgPtr"
|
||||
],
|
||||
"Args":[
|
||||
"const char*", "ThunkName",
|
||||
"uintptr_t", "ThunkFnPtr"
|
||||
"SHA256Sum", "ThunkNameHash"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -3310,6 +3331,15 @@
|
||||
]
|
||||
},
|
||||
|
||||
"F80LoadFCW": {
|
||||
"OpClass": "Vector",
|
||||
"HasSideEffects": true,
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"X80FCW"
|
||||
]
|
||||
},
|
||||
|
||||
"F80Add": {
|
||||
"OpClass": "Vector",
|
||||
"HasDest": true,
|
||||
@@ -3439,6 +3469,9 @@
|
||||
"SSANames": [
|
||||
"X80Src"
|
||||
],
|
||||
"Args": [
|
||||
"bool", "Truncate"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "Size"
|
||||
]
|
||||
|
||||
+16
-8
@@ -3,6 +3,8 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <iomanip>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define IROP_GETNAME_IMPL
|
||||
#define IROP_GETRAARGS_IMPL
|
||||
@@ -12,15 +14,21 @@ namespace FEXCore::IR {
|
||||
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false> const* IR, uint64_t Arg) {
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, const SHA256Sum &Arg) {
|
||||
*out << "sha256:";
|
||||
for(auto byte: Arg.data)
|
||||
*out << std::hex << std::setfill('0') << std::setw(2) << (unsigned int)byte;
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, uint64_t Arg) {
|
||||
*out << "#0x" << std::hex << Arg;
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false> const* IR, const char* Arg) {
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, const char* Arg) {
|
||||
*out << Arg;
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false> const* IR, CondClassType Arg) {
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, CondClassType Arg) {
|
||||
std::array<std::string, 22> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
@@ -49,7 +57,7 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false>
|
||||
*out << CondNames[Arg];
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false> const* IR, MemOffsetType Arg) {
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, MemOffsetType Arg) {
|
||||
std::array<std::string, 3> Names = {
|
||||
"SXTX",
|
||||
"UXTW",
|
||||
@@ -59,7 +67,7 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false>
|
||||
*out << Names[Arg];
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false> const* IR, RegisterClassType Arg) {
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, RegisterClassType Arg) {
|
||||
if (Arg == GPRClass.Val)
|
||||
*out << "GPR";
|
||||
else if (Arg == GPRFixedClass.Val)
|
||||
@@ -74,7 +82,7 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false>
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, IRListView<false> const* IR, OrderedNodeWrapper Arg, IR::RegisterAllocationData *RAData) {
|
||||
static void PrintArg(std::stringstream *out, IRListView const* IR, OrderedNodeWrapper Arg, IR::RegisterAllocationData *RAData) {
|
||||
auto [CodeNode, IROp] = IR->at(Arg)();
|
||||
|
||||
if (Arg.ID() == 0) {
|
||||
@@ -123,7 +131,7 @@ static void PrintArg(std::stringstream *out, IRListView<false> const* IR, Ordere
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false> const* IR, FEXCore::IR::FenceType Arg) {
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::FenceType Arg) {
|
||||
if (Arg == IR::Fence_Load) {
|
||||
*out << "Loads";
|
||||
}
|
||||
@@ -138,7 +146,7 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false>
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(std::stringstream *out, IRListView<false> const* IR, IR::RegisterAllocationData *RAData) {
|
||||
void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
int8_t CurrentIndent = 0;
|
||||
|
||||
+50
-1
@@ -81,6 +81,15 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, bool> DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint8_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE || Result > 1) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result != 0};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint16_t> DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
@@ -108,6 +117,46 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, int64_t> DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
int64_t Result = (int64_t)strtoull(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, IR::SHA256Sum> DecodeValue(std::string &Arg) {
|
||||
IR::SHA256Sum Result;
|
||||
|
||||
if (Arg.at(0) != 's' || Arg.at(1) != 'h' || Arg.at(2) != 'a' || Arg.at(3) != '2' || Arg.at(4) != '5' || Arg.at(5) != '6' || Arg.at(6) != ':')
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, Result};
|
||||
|
||||
auto GetDigit = [](const std::string &Arg, int pos, uint8_t *val) {
|
||||
auto chr = Arg.at(pos);
|
||||
if (chr >= '0' && chr <= '9') {
|
||||
*val = chr - '0';
|
||||
return true;
|
||||
} else if (chr >= 'a' && chr <= 'f') {
|
||||
*val = 10 + chr - 'a';
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < sizeof(Result.data); i++) {
|
||||
uint8_t high, low;
|
||||
if (!GetDigit(Arg, 7 + 2 * i + 0, &high) || !GetDigit(Arg, 7 + 2 * i + 1, &low)) {
|
||||
return {DecodeFailure::DECODE_INVALIDRANGE, Result};
|
||||
}
|
||||
Result.data[i] = high * 16 + low;
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::RegisterClassType> DecodeValue(std::string &Arg) {
|
||||
if (Arg == "GPR") {
|
||||
@@ -447,7 +496,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
if (!CheckPrintError(Def, Entry.first)) return false;
|
||||
if (!CheckPrintError(Def, CodeBlockCount.first)) return false;
|
||||
|
||||
IRHeader = _IRHeader(InvalidNode, Entry.second, CodeBlockCount.second, false);
|
||||
IRHeader = _IRHeader(InvalidNode, Entry.second, CodeBlockCount.second);
|
||||
}
|
||||
|
||||
SetWriteCursor(nullptr); // isolate the header from everything following
|
||||
|
||||
+14
-3
@@ -485,7 +485,7 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Header.Args[0]);
|
||||
|
||||
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8 && !Header->ShouldInterpret) {
|
||||
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
|
||||
|
||||
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, Op->Size, AddressHeader);
|
||||
|
||||
@@ -503,7 +503,7 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Header.Args[0]);
|
||||
|
||||
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8 && !Header->ShouldInterpret) {
|
||||
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
|
||||
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, Op->Size, AddressHeader);
|
||||
|
||||
Op->OffsetType = OffsetType;
|
||||
@@ -712,7 +712,9 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
uint64_t amt = __builtin_ctzl(Constant2);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto shift = IREmit->_Lshl(CurrentIR.GetNode(Op->Header.Args[0]), IREmit->_Constant(amt));
|
||||
shift.first->Header.Size = IROp->Size; // force Lshl to be the same size as the original Mul
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, shift);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
@@ -747,7 +749,7 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
}
|
||||
|
||||
// constant inlining
|
||||
if (!HeaderOp->ShouldInterpret && InlineConstants) {
|
||||
if (InlineConstants) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
switch(IROp->Op) {
|
||||
case OP_LSHR:
|
||||
@@ -852,6 +854,15 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(Constant));
|
||||
|
||||
Changed = true;
|
||||
} else {
|
||||
auto NewRIP = IREmit->GetOpHeader(Op->NewRIP);
|
||||
if (NewRIP->Op == OP_ENTRYPOINTOFFSET) {
|
||||
auto EO = NewRIP->C<IR::IROp_EntrypointOffset>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(EO->Offset, EO->Header.Size));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
+14
-1
@@ -33,7 +33,7 @@ namespace {
|
||||
std::vector<ContextMemberInfo> ClassificationInfo;
|
||||
};
|
||||
|
||||
constexpr static std::array<LastAccessType, 14> DefaultAccess = {
|
||||
constexpr static std::array<LastAccessType, 15> DefaultAccess = {
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_INVALID, // PAD
|
||||
@@ -48,6 +48,7 @@ namespace {
|
||||
ACCESS_INVALID, // PAD
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
};
|
||||
|
||||
static void ClassifyContextStruct(ContextInfo *ContextClassificationInfo) {
|
||||
@@ -190,6 +191,16 @@ namespace {
|
||||
});
|
||||
}
|
||||
|
||||
// FCW
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, FCW),
|
||||
sizeof(FEXCore::Core::CPUState::FCW),
|
||||
},
|
||||
DefaultAccess[14],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
size_t ClassifiedStructSize{};
|
||||
ContextClassificationInfo->Lookup.reserve(sizeof(FEXCore::Core::CPUState));
|
||||
for (auto &it : *ContextClassification) {
|
||||
@@ -250,6 +261,8 @@ namespace {
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[13]);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[14]);
|
||||
}
|
||||
|
||||
struct BlockInfo {
|
||||
|
||||
@@ -77,7 +77,7 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
|
||||
// Zero is always zero(invalid)
|
||||
OldToNewRemap[0].NodeID = 0;
|
||||
auto LocalHeaderOp = LocalBuilder._IRHeader(OrderedNodeWrapper::WrapOffset(0).GetNode(ListBegin), HeaderOp->Entry, HeaderOp->BlockCount, HeaderOp->ShouldInterpret);
|
||||
auto LocalHeaderOp = LocalBuilder._IRHeader(OrderedNodeWrapper::WrapOffset(0).GetNode(ListBegin), HeaderOp->Entry, HeaderOp->BlockCount);
|
||||
OldToNewRemap[CurrentIR.GetID(HeaderNode)].NodeID = LocalIR.GetID(LocalHeaderOp.Node);
|
||||
|
||||
{
|
||||
|
||||
@@ -51,10 +51,12 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
|
||||
IR::RegisterAllocationData * RAData{};
|
||||
if (Manager->HasRAPass() && !HeaderOp->ShouldInterpret) {
|
||||
if (Manager->HasRAPass()) {
|
||||
RAData = Manager->GetRAPass() ? Manager->GetRAPass()->GetAllocationData() : nullptr;
|
||||
}
|
||||
|
||||
NodeIsLive.Set(1); // IRHEADER
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
@@ -270,7 +272,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
Out << "Warnings:" << std::endl << Warnings.str() << std::endl;
|
||||
}
|
||||
|
||||
LogMan::Msg::E("%s", Out.str().c_str());
|
||||
fprintf(stderr, "%s", Out.str().c_str());
|
||||
}
|
||||
|
||||
return false;
|
||||
|
||||
@@ -258,6 +258,8 @@ namespace {
|
||||
Graph->AllocData.reset();
|
||||
Graph->AllocData.reset((FEXCore::IR::RegisterAllocationData*)malloc(FEXCore::IR::RegisterAllocationData::Size(NodeCount)));
|
||||
memset(&Graph->AllocData->Map[0], INVALID_REGCLASS.Raw, NodeCount);
|
||||
Graph->AllocData->MapCount = NodeCount;
|
||||
Graph->AllocData->IsShared = false; // not shared by default
|
||||
Graph->NodeCount = NodeCount;
|
||||
}
|
||||
|
||||
@@ -302,7 +304,7 @@ namespace {
|
||||
}
|
||||
#endif
|
||||
|
||||
FEXCore::IR::RegisterClassType GetRegClassFromNode(FEXCore::IR::IRListView<false> *IR, FEXCore::IR::IROp_Header *IROp) {
|
||||
FEXCore::IR::RegisterClassType GetRegClassFromNode(FEXCore::IR::IRListView *IR, FEXCore::IR::IROp_Header *IROp) {
|
||||
using namespace FEXCore;
|
||||
|
||||
FEXCore::IR::RegisterClassType Class = IR::GetRegClass(IROp->Op);
|
||||
@@ -356,7 +358,7 @@ namespace {
|
||||
};
|
||||
|
||||
// Walk the IR and set the node classes
|
||||
void FindNodeClasses(RegisterGraph *Graph, FEXCore::IR::IRListView<false> *IR) {
|
||||
void FindNodeClasses(RegisterGraph *Graph, FEXCore::IR::IRListView *IR) {
|
||||
for (auto [CodeNode, IROp] : IR->GetAllCode()) {
|
||||
// If the destination hasn't yet been set then set it now
|
||||
if (IROp->HasDest) {
|
||||
@@ -410,14 +412,14 @@ namespace FEXCore::IR {
|
||||
std::unordered_map<uint32_t, BlockInterferences> LocalBlockInterferences;
|
||||
BlockInterferences GlobalBlockInterferences;
|
||||
|
||||
void CalculateLiveRange(FEXCore::IR::IRListView<false> *IR);
|
||||
void OptimizeStaticRegisters(FEXCore::IR::IRListView<false> *IR);
|
||||
void CalculateBlockInterferences(FEXCore::IR::IRListView<false> *IR);
|
||||
void CalculateBlockNodeInterference(FEXCore::IR::IRListView<false> *IR);
|
||||
void CalculateNodeInterference(FEXCore::IR::IRListView<false> *IR);
|
||||
void CalculateLiveRange(FEXCore::IR::IRListView *IR);
|
||||
void OptimizeStaticRegisters(FEXCore::IR::IRListView *IR);
|
||||
void CalculateBlockInterferences(FEXCore::IR::IRListView *IR);
|
||||
void CalculateBlockNodeInterference(FEXCore::IR::IRListView *IR);
|
||||
void CalculateNodeInterference(FEXCore::IR::IRListView *IR);
|
||||
void AllocateVirtualRegisters();
|
||||
void CalculatePredecessors(FEXCore::IR::IRListView<false> *IR);
|
||||
void RecursiveLiveRangeExpansion(FEXCore::IR::IRListView<false> *IR, uint32_t Node, uint32_t DefiningBlockID, LiveRange *LiveRange, const std::unordered_set<uint32_t> &Predecessors, std::unordered_set<uint32_t> &VisitedPredecessors);
|
||||
void CalculatePredecessors(FEXCore::IR::IRListView *IR);
|
||||
void RecursiveLiveRangeExpansion(FEXCore::IR::IRListView *IR, uint32_t Node, uint32_t DefiningBlockID, LiveRange *LiveRange, const std::unordered_set<uint32_t> &Predecessors, std::unordered_set<uint32_t> &VisitedPredecessors);
|
||||
|
||||
FEXCore::IR::AllNodesIterator FindFirstUse(FEXCore::IR::IREmitter *IREmit, FEXCore::IR::OrderedNode* Node, FEXCore::IR::AllNodesIterator Begin, FEXCore::IR::AllNodesIterator End);
|
||||
FEXCore::IR::AllNodesIterator FindLastUseBefore(FEXCore::IR::IREmitter *IREmit, FEXCore::IR::OrderedNode* Node, FEXCore::IR::AllNodesIterator Begin, FEXCore::IR::AllNodesIterator End);
|
||||
@@ -468,7 +470,7 @@ namespace FEXCore::IR {
|
||||
return std::move(Graph->AllocData);
|
||||
}
|
||||
|
||||
void ConstrainedRAPass::RecursiveLiveRangeExpansion(FEXCore::IR::IRListView<false> *IR, uint32_t Node, uint32_t DefiningBlockID, LiveRange *LiveRange, const std::unordered_set<uint32_t> &Predecessors, std::unordered_set<uint32_t> &VisitedPredecessors) {
|
||||
void ConstrainedRAPass::RecursiveLiveRangeExpansion(FEXCore::IR::IRListView *IR, uint32_t Node, uint32_t DefiningBlockID, LiveRange *LiveRange, const std::unordered_set<uint32_t> &Predecessors, std::unordered_set<uint32_t> &VisitedPredecessors) {
|
||||
for (auto PredecessorId: Predecessors) {
|
||||
if (DefiningBlockID != PredecessorId && !VisitedPredecessors.contains(PredecessorId)) {
|
||||
// do the magic
|
||||
@@ -491,7 +493,7 @@ namespace FEXCore::IR {
|
||||
}
|
||||
}
|
||||
|
||||
void ConstrainedRAPass::CalculateLiveRange(FEXCore::IR::IRListView<false> *IR) {
|
||||
void ConstrainedRAPass::CalculateLiveRange(FEXCore::IR::IRListView *IR) {
|
||||
using namespace FEXCore;
|
||||
size_t Nodes = IR->GetSSACount();
|
||||
LiveRanges.clear();
|
||||
@@ -535,6 +537,8 @@ namespace FEXCore::IR {
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) continue;
|
||||
if (IR->GetOp<IROp_Header>(IROp->Args[i])->Op == OP_INLINECONSTANT) continue;
|
||||
if (IR->GetOp<IROp_Header>(IROp->Args[i])->Op == OP_INLINEENTRYPOINTOFFSET) continue;
|
||||
if (IR->GetOp<IROp_Header>(IROp->Args[i])->Op == OP_IRHEADER) continue;
|
||||
uint32_t ArgNode = IROp->Args[i].ID();
|
||||
LogMan::Throw::A(LiveRanges[ArgNode].Begin != ~0U, "%%ssa%d used by %%ssa%d before defined?", ArgNode, Node);
|
||||
|
||||
@@ -579,7 +583,7 @@ namespace FEXCore::IR {
|
||||
}
|
||||
}
|
||||
|
||||
void ConstrainedRAPass::OptimizeStaticRegisters(FEXCore::IR::IRListView<false> *IR) {
|
||||
void ConstrainedRAPass::OptimizeStaticRegisters(FEXCore::IR::IRListView *IR) {
|
||||
|
||||
// Helpers
|
||||
|
||||
@@ -699,6 +703,8 @@ namespace FEXCore::IR {
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) continue;
|
||||
if (IR->GetOp<IROp_Header>(IROp->Args[i])->Op == OP_INLINECONSTANT) continue;
|
||||
if (IR->GetOp<IROp_Header>(IROp->Args[i])->Op == OP_INLINEENTRYPOINTOFFSET) continue;
|
||||
if (IR->GetOp<IROp_Header>(IROp->Args[i])->Op == OP_IRHEADER) continue;
|
||||
uint32_t ArgNode = IROp->Args[i].ID();
|
||||
|
||||
// ACCESSED after write, let's not SRA this one
|
||||
@@ -790,7 +796,7 @@ namespace FEXCore::IR {
|
||||
}
|
||||
}
|
||||
|
||||
void ConstrainedRAPass::CalculateBlockInterferences(FEXCore::IR::IRListView<false> *IR) {
|
||||
void ConstrainedRAPass::CalculateBlockInterferences(FEXCore::IR::IRListView *IR) {
|
||||
using namespace FEXCore;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
@@ -818,7 +824,7 @@ namespace FEXCore::IR {
|
||||
}
|
||||
}
|
||||
|
||||
void ConstrainedRAPass::CalculateBlockNodeInterference(FEXCore::IR::IRListView<false> *IR) {
|
||||
void ConstrainedRAPass::CalculateBlockNodeInterference(FEXCore::IR::IRListView *IR) {
|
||||
#if 0
|
||||
auto AddInterference = [&](uint32_t Node1, uint32_t Node2) {
|
||||
RegisterNode *Node = &Graph->Nodes[Node1];
|
||||
@@ -877,7 +883,7 @@ namespace FEXCore::IR {
|
||||
#endif
|
||||
}
|
||||
|
||||
void ConstrainedRAPass::CalculateNodeInterference(FEXCore::IR::IRListView<false> *IR) {
|
||||
void ConstrainedRAPass::CalculateNodeInterference(FEXCore::IR::IRListView *IR) {
|
||||
auto AddInterference = [this](uint32_t Node1, uint32_t Node2) {
|
||||
RegisterNode *Node = &Graph->Nodes[Node1];
|
||||
Node->Interferences.Append(Node2);
|
||||
@@ -1392,7 +1398,6 @@ namespace FEXCore::IR {
|
||||
//LogMan::Throw::A(InterferenceRegisterNode->Head.RegAndClass.Reg != INVALID_REG, "Interference node never assigned a register?");
|
||||
LogMan::Throw::A(InterferenceRegClass != ~0U, "Interference node never assigned a register class?");
|
||||
LogMan::Throw::A(InterferenceRegisterNode->Head.PhiPartner == nullptr, "We don't support spilling PHI nodes currently");
|
||||
LogMan::Throw::A(InterferenceRegClass == RegClass, "Class doesn't match");
|
||||
|
||||
// This is the op that we need to dump
|
||||
auto [InterferenceOrderedNode, InterferenceIROp] = IR.at(InterferenceNode)();
|
||||
@@ -1478,7 +1483,7 @@ namespace FEXCore::IR {
|
||||
}
|
||||
|
||||
|
||||
void ConstrainedRAPass::CalculatePredecessors(FEXCore::IR::IRListView<false> *IR) {
|
||||
void ConstrainedRAPass::CalculatePredecessors(FEXCore::IR::IRListView *IR) {
|
||||
Graph->BlockPredecessors.clear();
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : IR->GetBlocks()) {
|
||||
@@ -1502,9 +1507,6 @@ namespace FEXCore::IR {
|
||||
auto IR = IREmit->ViewIR();
|
||||
|
||||
auto HeaderOp = IR.GetHeader();
|
||||
if (HeaderOp->ShouldInterpret) {
|
||||
return false;
|
||||
}
|
||||
|
||||
SpillSlotCount = 0;
|
||||
Graph->SpillStack.clear();
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
template<bool>
|
||||
class IRListView;
|
||||
|
||||
class RegisterAllocationPass : public FEXCore::IR::Pass {
|
||||
|
||||
@@ -44,9 +44,6 @@ bool IsStaticAllocFpr(uint32_t Offset, RegisterClassType Class, bool AllowGpr) {
|
||||
bool StaticRegisterAllocationPass::Run(IREmitter *IREmit) {
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
if (CurrentIR.GetHeader()->ShouldInterpret)
|
||||
return false;
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
@@ -81,6 +81,8 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint32_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) continue;
|
||||
if (CurrentIR.GetOp<IROp_Header>(IROp->Args[i])->Op == OP_IRHEADER) continue;
|
||||
|
||||
OrderedNodeWrapper Arg = IROp->Args[i];
|
||||
|
||||
// We must ensure domininance of all SSA arguments
|
||||
|
||||
Vendored
+2
-2
@@ -79,10 +79,10 @@ This is an intrusive allocator that is used by the `OpDispatchBuilder` for stori
|
||||
|
||||
### OpDispatchBuilder
|
||||
OpDispatchBuilder provides two routines for handling the IR outside of the class
|
||||
* `IRListView<false> ViewIR();`
|
||||
* `IRListView ViewIR();`
|
||||
* Returns a wrapper container class the allows you to view the IR. This doesn't take ownership of the IR data.
|
||||
* If the OpDispatcherBuilder changes its IR then changes are also visible to this class
|
||||
* `IRListView<true> *CreateIRCopy()`
|
||||
* `IRListView *CreateIRCopy()`
|
||||
* As the name says, it creates a new copy of the IR that is in the OpDispatchBuilder
|
||||
* Copying the IR only copies the memory used and doesn't have any free space for optimizations after this copy operation
|
||||
* Useful for tiered recompilers, AOT, and offline analysis
|
||||
|
||||
@@ -34,6 +34,8 @@ namespace FEXCore::Config {
|
||||
CONFIG_INTERPRETER_INSTALLED,
|
||||
CONFIG_APP_FILENAME,
|
||||
CONFIG_DEBUG_DISABLE_OPTIMIZATION_PASSES,
|
||||
CONFIG_AOTIR_GENERATE,
|
||||
CONFIG_AOTIR_LOAD
|
||||
};
|
||||
|
||||
enum ConfigCore {
|
||||
@@ -42,6 +44,12 @@ namespace FEXCore::Config {
|
||||
CONFIG_CUSTOM,
|
||||
};
|
||||
|
||||
enum ConfigSMCChecks {
|
||||
CONFIG_SMC_NONE,
|
||||
CONFIG_SMC_MMAN,
|
||||
CONFIG_SMC_FULL,
|
||||
};
|
||||
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config);
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, std::string const &Config);
|
||||
uint64_t GetConfig(FEXCore::Context::Context *CTX, ConfigOption Option);
|
||||
|
||||
+1
-2
@@ -5,7 +5,6 @@
|
||||
namespace FEXCore {
|
||||
|
||||
namespace IR {
|
||||
template<bool Copy>
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
}
|
||||
@@ -45,7 +44,7 @@ class LLVMCore;
|
||||
* @return An executable function pointer that is theoretically compiled from this point.
|
||||
* Is actually a function pointer of type `void (FEXCore::Core::ThreadState *Thread)
|
||||
*/
|
||||
virtual void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) = 0;
|
||||
virtual void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) = 0;
|
||||
|
||||
/**
|
||||
* @brief Function for mapping memory in to the CPUBackend's visible space. Allows setting up virtual mappings if required
|
||||
|
||||
+10
@@ -6,6 +6,10 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
|
||||
#include <istream>
|
||||
#include <ostream>
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
}
|
||||
@@ -229,4 +233,10 @@ namespace FEXCore::Context {
|
||||
void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation);
|
||||
void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler);
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf);
|
||||
|
||||
void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name);
|
||||
void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length);
|
||||
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::istream>(const std::string&)> CacheReader);
|
||||
bool WriteAOTIR(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
|
||||
void FlushCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length);
|
||||
}
|
||||
@@ -22,6 +22,7 @@ namespace FEXCore::Core {
|
||||
struct {
|
||||
uint32_t base;
|
||||
} gdt[32];
|
||||
uint16_t FCW;
|
||||
};
|
||||
static_assert(offsetof(CPUState, xmm) % 16 == 0, "xmm needs to be 128bit aligned!");
|
||||
|
||||
|
||||
@@ -61,6 +61,14 @@ namespace FEXCore::Core {
|
||||
SIGNALEVENT_RETURN,
|
||||
};
|
||||
|
||||
struct LocalIREntry {
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
std::unique_ptr<FEXCore::IR::IRListView, FEXCore::IR::IRListViewDeleter> IR;
|
||||
std::unique_ptr<FEXCore::IR::RegisterAllocationData, FEXCore::IR::RegisterAllocationDataDeleter> RAData;
|
||||
std::unique_ptr<FEXCore::Core::DebugData> DebugData;
|
||||
};
|
||||
|
||||
struct InternalThreadState {
|
||||
FEXCore::Core::ThreadState State;
|
||||
|
||||
@@ -76,9 +84,7 @@ namespace FEXCore::Core {
|
||||
std::unique_ptr<FEXCore::CPU::CPUBackend> CPUBackend;
|
||||
std::unique_ptr<FEXCore::LookupCache> LookupCache;
|
||||
|
||||
std::unordered_map<uint64_t, std::unique_ptr<FEXCore::IR::IRListView<true>>> IRLists;
|
||||
std::unordered_map<uint64_t, std::unique_ptr<FEXCore::IR::RegisterAllocationData, FEXCore::IR::RegisterAllocationDataDeleter>> RALists;
|
||||
std::unordered_map<uint64_t, std::unique_ptr<FEXCore::Core::DebugData>> DebugData;
|
||||
std::unordered_map<uint64_t, LocalIREntry> LocalIRCache;
|
||||
|
||||
std::unique_ptr<FEXCore::Frontend::Decoder> FrontendDecoder;
|
||||
std::unique_ptr<FEXCore::IR::PassManager> PassManager;
|
||||
|
||||
+7
-2
@@ -2,6 +2,7 @@
|
||||
#include <array>
|
||||
#include <cassert>
|
||||
#include <cstdint>
|
||||
#include <string.h>
|
||||
#include <sstream>
|
||||
#include <tuple>
|
||||
|
||||
@@ -325,6 +326,11 @@ struct FenceType final {
|
||||
constexpr bool operator!=(FenceType const &rhs) const { return !operator==(rhs); }
|
||||
};
|
||||
|
||||
struct SHA256Sum final {
|
||||
uint8_t data[32];
|
||||
bool operator<(SHA256Sum const &rhs) const { return memcmp(data, rhs.data, sizeof(data)) < 0; }
|
||||
};
|
||||
|
||||
class NodeIterator;
|
||||
|
||||
/* This iterator can be used to step though nodes.
|
||||
@@ -456,11 +462,10 @@ public:
|
||||
}
|
||||
};
|
||||
|
||||
template<bool>
|
||||
class IRListView;
|
||||
class IREmitter;
|
||||
|
||||
void Dump(std::stringstream *out, IRListView<false> const* IR, IR::RegisterAllocationData *RAData);
|
||||
void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData);
|
||||
IREmitter* Parse(std::istream *in);
|
||||
|
||||
template<typename Type>
|
||||
|
||||
+2
-2
@@ -21,8 +21,8 @@ friend class FEXCore::IR::PassManager;
|
||||
ResetWorkingList();
|
||||
}
|
||||
|
||||
IRListView<false> ViewIR() { return IRListView<false>(&Data, &ListData); }
|
||||
IRListView<true> *CreateIRCopy() { return new IRListView<true>(&Data, &ListData); }
|
||||
IRListView ViewIR() { return IRListView(&Data, &ListData, false); }
|
||||
IRListView *CreateIRCopy() { return new IRListView(&Data, &ListData, true); }
|
||||
void ResetWorkingList();
|
||||
|
||||
/**
|
||||
|
||||
+43
-10
@@ -8,6 +8,8 @@
|
||||
#include <cstring>
|
||||
#include <tuple>
|
||||
#include <vector>
|
||||
#include <istream>
|
||||
#include <ostream>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
/**
|
||||
@@ -62,17 +64,16 @@ class IntrusiveAllocator final {
|
||||
uintptr_t Data;
|
||||
};
|
||||
|
||||
template<bool Copy>
|
||||
class IRListView final {
|
||||
public:
|
||||
IRListView() = delete;
|
||||
IRListView(IRListView<Copy> &&) = delete;
|
||||
IRListView(IRListView &&) = delete;
|
||||
|
||||
IRListView(IntrusiveAllocator *Data, IntrusiveAllocator *List) {
|
||||
IRListView(IntrusiveAllocator *Data, IntrusiveAllocator *List, bool _IsCopy) : IsCopy(_IsCopy) {
|
||||
DataSize = Data->Size();
|
||||
ListSize = List->Size();
|
||||
|
||||
if (Copy) {
|
||||
if (IsCopy) {
|
||||
IRData = malloc(DataSize + ListSize);
|
||||
ListData = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(IRData) + DataSize);
|
||||
memcpy(IRData, reinterpret_cast<void*>(Data->Begin()), DataSize);
|
||||
@@ -85,24 +86,46 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
IRListView<true>(IRListView<true> *Old) {
|
||||
IRListView(IRListView *Old, bool _IsCopy) : IsCopy(_IsCopy) {
|
||||
DataSize = Old->DataSize;
|
||||
ListSize = Old->ListSize;
|
||||
if (IsCopy) {
|
||||
IRData = malloc(DataSize + ListSize);
|
||||
ListData = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(IRData) + DataSize);
|
||||
memcpy(IRData, Old->IRData, DataSize);
|
||||
memcpy(ListData, Old->ListData, ListSize);
|
||||
} else {
|
||||
IRData = Old->IRData;
|
||||
ListData = Old->ListData;
|
||||
}
|
||||
}
|
||||
|
||||
IRListView(std::istream& stream) : IsCopy(true) {
|
||||
stream.read((char*)&DataSize, sizeof(DataSize));
|
||||
stream.read((char*)&ListSize, sizeof(ListSize));
|
||||
|
||||
IRData = malloc(DataSize + ListSize);
|
||||
ListData = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(IRData) + DataSize);
|
||||
memcpy(IRData, Old->IRData, DataSize);
|
||||
memcpy(ListData, Old->ListData, ListSize);
|
||||
stream.read((char*)IRData, DataSize);
|
||||
stream.read((char*)ListData, ListSize);
|
||||
}
|
||||
|
||||
~IRListView() {
|
||||
if (Copy) {
|
||||
if (IsCopy) {
|
||||
free (IRData);
|
||||
// ListData is just offset from IRData
|
||||
}
|
||||
}
|
||||
|
||||
IRListView<true> *CreateCopy() {
|
||||
return new IRListView<true>(this);
|
||||
void Serialize(std::ostream& stream) {
|
||||
stream.write((char*)&DataSize, sizeof(DataSize));
|
||||
stream.write((char*)&ListSize, sizeof(ListSize));
|
||||
stream.write((char*)IRData, DataSize);
|
||||
stream.write((char*)ListData, ListSize);
|
||||
}
|
||||
|
||||
IRListView *CreateCopy() {
|
||||
return new IRListView(this, true);
|
||||
}
|
||||
|
||||
uintptr_t const GetData() const { return reinterpret_cast<uintptr_t>(IRData); }
|
||||
@@ -149,6 +172,7 @@ public:
|
||||
return Wrapper.GetNode(GetListData());
|
||||
}
|
||||
|
||||
bool IsShared {false};
|
||||
private:
|
||||
struct BlockRange {
|
||||
using iterator = NodeIterator;
|
||||
@@ -259,6 +283,15 @@ private:
|
||||
void *ListData;
|
||||
size_t DataSize;
|
||||
size_t ListSize;
|
||||
bool IsCopy;
|
||||
};
|
||||
|
||||
struct IRListViewDeleter {
|
||||
void operator()(IRListView* r) {
|
||||
if (!r->IsShared) {
|
||||
delete r;
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
@@ -27,16 +27,11 @@ union PhysicalRegister {
|
||||
|
||||
static_assert(sizeof(PhysicalRegister) == 1);
|
||||
|
||||
class RegisterAllocationData;
|
||||
struct RegisterAllocationDataDeleter {
|
||||
void operator()(RegisterAllocationData* r) {
|
||||
free(r);
|
||||
}
|
||||
};
|
||||
|
||||
class RegisterAllocationData {
|
||||
public:
|
||||
uint32_t SpillSlotCount {};
|
||||
uint32_t MapCount {};
|
||||
bool IsShared {false};
|
||||
PhysicalRegister Map[0];
|
||||
|
||||
PhysicalRegister GetNodeRegister(uint32_t Node) const {
|
||||
@@ -49,4 +44,12 @@ class RegisterAllocationData {
|
||||
}
|
||||
};
|
||||
|
||||
struct RegisterAllocationDataDeleter {
|
||||
void operator()(RegisterAllocationData* r) {
|
||||
if (!r->IsShared) {
|
||||
free(r);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
}
|
||||
@@ -66,11 +66,11 @@ namespace FEX::ArgLoader {
|
||||
.help("Number of physical hardware threads to tell the process we have")
|
||||
.set_default(1);
|
||||
|
||||
CPUGroup.add_option("--smc-full-checks")
|
||||
CPUGroup.add_option("--smc-checks")
|
||||
.dest("SMCChecks")
|
||||
.action("store_true")
|
||||
.help("Checks code for modification before execution. Slow.")
|
||||
.set_default(false);
|
||||
.choices({"none", "mman", "full"})
|
||||
.help("Checks code for modification before execution.\n\tnone: No checks\n\tmman: Invalidate on mmap, mprotect, munmap\n\tfull: Validate code before every run (slow)")
|
||||
.set_default("mman");
|
||||
|
||||
CPUGroup.add_option("--unsafe-no-tso")
|
||||
.dest("TSOEnabled")
|
||||
@@ -114,6 +114,18 @@ namespace FEX::ArgLoader {
|
||||
.help("Disables optimization passes for debugging")
|
||||
.choices({"0"});
|
||||
|
||||
EmulationGroup.add_option("--aotir-capture")
|
||||
.dest("AOTIRCapture")
|
||||
.help("Captures IR and generates an AOTIR cache for the loaded executable and libs")
|
||||
.action("store_true")
|
||||
.set_default(false);
|
||||
|
||||
EmulationGroup.add_option("--aotir-load")
|
||||
.dest("AOTIRLoad")
|
||||
.help("Loads an AOTIR cache for the loaded executable")
|
||||
.action("store_true")
|
||||
.set_default(false);
|
||||
|
||||
Parser.add_option_group(EmulationGroup);
|
||||
}
|
||||
{
|
||||
@@ -217,8 +229,13 @@ namespace FEX::ArgLoader {
|
||||
}
|
||||
|
||||
if (Options.is_set_by_user("SMCChecks")) {
|
||||
bool SMCChecks = Options.get("SMCChecks");
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_SMC_CHECKS, std::to_string(SMCChecks));
|
||||
auto SMCChecks = Options["SMCChecks"];
|
||||
if (SMCChecks == "none")
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_SMC_CHECKS, "0");
|
||||
else if (SMCChecks == "mman")
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_SMC_CHECKS, "1");
|
||||
else if (SMCChecks == "full")
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_SMC_CHECKS, "2");
|
||||
}
|
||||
if (Options.is_set_by_user("AbiLocalFlags")) {
|
||||
bool AbiLocalFlags = Options.get("AbiLocalFlags");
|
||||
@@ -249,6 +266,16 @@ namespace FEX::ArgLoader {
|
||||
if (Options.is_set_by_user("O0")) {
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_DEBUG_DISABLE_OPTIMIZATION_PASSES, std::to_string(true));
|
||||
}
|
||||
|
||||
if (Options.is_set_by_user("AOTIRCapture")) {
|
||||
bool AOTIRCapture = Options.get("AOTIRCapture");
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_AOTIR_GENERATE, std::to_string(AOTIRCapture));
|
||||
}
|
||||
|
||||
if (Options.is_set_by_user("AOTIRLoad")) {
|
||||
bool AOTIRLoad = Options.get("AOTIRLoad");
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_AOTIR_LOAD, std::to_string(AOTIRLoad));
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
|
||||
@@ -188,6 +188,8 @@ namespace FEX::Config {
|
||||
{FEXCore::Config::ConfigOption::CONFIG_ABI_LOCAL_FLAGS, "ABILocalFlags"},
|
||||
{FEXCore::Config::ConfigOption::CONFIG_ABI_NO_PF, "ABINoPF"},
|
||||
{FEXCore::Config::ConfigOption::CONFIG_DEBUG_DISABLE_OPTIMIZATION_PASSES, "O0"},
|
||||
{FEXCore::Config::ConfigOption::CONFIG_AOTIR_GENERATE, "AOTIRCapture"},
|
||||
{FEXCore::Config::ConfigOption::CONFIG_AOTIR_LOAD, "AOTIRLoad"},
|
||||
}};
|
||||
|
||||
|
||||
@@ -235,6 +237,8 @@ namespace FEX::Config {
|
||||
{"ABILocalFlags", FEXCore::Config::ConfigOption::CONFIG_ABI_LOCAL_FLAGS},
|
||||
{"AbiNoPF", FEXCore::Config::ConfigOption::CONFIG_ABI_NO_PF},
|
||||
{"O0", FEXCore::Config::ConfigOption::CONFIG_DEBUG_DISABLE_OPTIMIZATION_PASSES},
|
||||
{"AOTIRCapture", FEXCore::Config::ConfigOption::CONFIG_AOTIR_GENERATE},
|
||||
{"AOTIRLoad", FEXCore::Config::ConfigOption::CONFIG_AOTIR_LOAD},
|
||||
}};
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
@@ -307,7 +311,7 @@ namespace FEX::Config {
|
||||
}
|
||||
};
|
||||
|
||||
static const std::array<std::pair<std::string, FEXCore::Config::ConfigOption>, 18> ConfigLookup = {{
|
||||
static const std::array<std::pair<std::string, FEXCore::Config::ConfigOption>, 20> ConfigLookup = {{
|
||||
{"FEX_CORE", FEXCore::Config::ConfigOption::CONFIG_DEFAULTCORE},
|
||||
{"FEX_MAXINST", FEXCore::Config::ConfigOption::CONFIG_MAXBLOCKINST},
|
||||
{"FEX_SINGLESTEP", FEXCore::Config::ConfigOption::CONFIG_SINGLESTEP},
|
||||
@@ -326,6 +330,8 @@ namespace FEX::Config {
|
||||
{"FEX_ABINOPF", FEXCore::Config::ConfigOption::CONFIG_ABI_NO_PF},
|
||||
{"FEX_BREAK", FEXCore::Config::ConfigOption::CONFIG_BREAK_ON_FRONTEND},
|
||||
{"FEX_DUMP_GPRS", FEXCore::Config::ConfigOption::CONFIG_DUMP_GPRS},
|
||||
{"FEX_AOT_GENERATE", FEXCore::Config::ConfigOption::CONFIG_AOTIR_GENERATE},
|
||||
{"FEX_AOT_LOAD", FEXCore::Config::ConfigOption::CONFIG_AOTIR_LOAD},
|
||||
}};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
@@ -30,7 +30,7 @@ namespace HostFactory {
|
||||
explicit HostCore(FEXCore::Context::Context* CTX, FEXCore::Core::ThreadState *Thread, bool Fallback);
|
||||
~HostCore() override;
|
||||
std::string GetName() override { return "Host Core"; }
|
||||
void* CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
void* CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void *HostPtr, uint64_t VirtualGuestPtr, uint64_t Size) override {
|
||||
return HostPtr;
|
||||
@@ -170,7 +170,7 @@ namespace HostFactory {
|
||||
ready();
|
||||
}
|
||||
|
||||
void* HostCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void* HostCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
|
||||
@@ -16,6 +16,8 @@
|
||||
#include <string>
|
||||
#include <unistd.h>
|
||||
#include <vector>
|
||||
#include <fstream>
|
||||
#include <filesystem>
|
||||
|
||||
namespace {
|
||||
static bool SilentLog;
|
||||
@@ -203,9 +205,11 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Config::Value<std::string> OutputLog{FEXCore::Config::CONFIG_OUTPUTLOG, "stderr"};
|
||||
FEXCore::Config::Value<std::string> DumpIR{FEXCore::Config::CONFIG_DUMPIR, "no"};
|
||||
FEXCore::Config::Value<bool> TSOEnabledConfig{FEXCore::Config::CONFIG_TSO_ENABLED, true};
|
||||
FEXCore::Config::Value<bool> SMCChecksConfig{FEXCore::Config::CONFIG_SMC_CHECKS, false};
|
||||
FEXCore::Config::Value<uint8_t> SMCChecksConfig{FEXCore::Config::CONFIG_SMC_CHECKS, FEXCore::Config::CONFIG_SMC_MMAN};
|
||||
FEXCore::Config::Value<bool> ABILocalFlags{FEXCore::Config::CONFIG_ABI_LOCAL_FLAGS, false};
|
||||
FEXCore::Config::Value<bool> AbiNoPF{FEXCore::Config::CONFIG_ABI_NO_PF, false};
|
||||
FEXCore::Config::Value<bool> AOTIRCapture{FEXCore::Config::CONFIG_AOTIR_GENERATE, false};
|
||||
FEXCore::Config::Value<bool> AOTIRLoad{FEXCore::Config::CONFIG_AOTIR_LOAD, false};
|
||||
|
||||
::SilentLog = SilentLog();
|
||||
|
||||
@@ -260,6 +264,8 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Config::SetConfig(CTX, FEXCore::Config::CONFIG_DUMPIR, DumpIR());
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_FILENAME, std::filesystem::canonical(Program));
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS64BIT_MODE, Loader.Is64BitMode() ? "1" : "0");
|
||||
FEXCore::Config::SetConfig(CTX, FEXCore::Config::CONFIG_AOTIR_GENERATE, AOTIRCapture());
|
||||
FEXCore::Config::SetConfig(CTX, FEXCore::Config::CONFIG_AOTIR_LOAD, AOTIRLoad());
|
||||
|
||||
std::unique_ptr<FEX::HLE::SignalDelegator> SignalDelegation = std::make_unique<FEX::HLE::SignalDelegator>();
|
||||
std::unique_ptr<FEX::HLE::SyscallHandler> SyscallHandler{
|
||||
@@ -287,8 +293,41 @@ int main(int argc, char **argv, char **const envp) {
|
||||
});
|
||||
}
|
||||
|
||||
if (AOTIRLoad() || AOTIRCapture()) {
|
||||
LogMan::Msg::I("Warning: AOTIR is experimental, and might lead to crashes. Capture doesn't work with programs that fork.");
|
||||
}
|
||||
|
||||
FEXCore::Context::SetAOTIRLoader(CTX, [](const std::string &fileid) -> std::unique_ptr<std::istream> {
|
||||
auto filepath = std::filesystem::path(getenv("HOME")) / ".fex-emu" / "aotir" / fileid;
|
||||
|
||||
return std::make_unique<std::ifstream>(filepath, std::ios::in | std::ios::binary);
|
||||
});
|
||||
|
||||
FEXCore::Context::RunUntilExit(CTX);
|
||||
|
||||
if (AOTIRCapture()) {
|
||||
std::filesystem::create_directories(std::filesystem::path(getenv("HOME")) / ".fex-emu" / "aotir");
|
||||
|
||||
auto WroteCache = FEXCore::Context::WriteAOTIR(CTX, [](const std::string& fileid) -> std::unique_ptr<std::ostream> {
|
||||
auto filepath = std::filesystem::path(getenv("HOME")) / ".fex-emu" / "aotir" / fileid;
|
||||
auto AOTWrite = std::make_unique<std::ofstream>(filepath, std::ios::out | std::ios::binary);
|
||||
if (*AOTWrite) {
|
||||
std::filesystem::resize_file(filepath, 0);
|
||||
AOTWrite->seekp(0);
|
||||
LogMan::Msg::I("AOTIR: Storing %s", fileid.c_str());
|
||||
} else {
|
||||
LogMan::Msg::I("AOTIR: Failed to store %s", fileid.c_str());
|
||||
}
|
||||
return AOTWrite;
|
||||
});
|
||||
|
||||
if (WroteCache) {
|
||||
LogMan::Msg::I("AOTIR Cache Stored");
|
||||
} else {
|
||||
LogMan::Msg::E("AOTIR Cache Store Failed");
|
||||
}
|
||||
}
|
||||
|
||||
auto ProgramStatus = FEXCore::Context::GetProgramStatus(CTX);
|
||||
|
||||
SyscallHandler.reset();
|
||||
|
||||
@@ -79,7 +79,7 @@ namespace FEX::EmulatedFile {
|
||||
|
||||
uint32_t Family = info.FamilyID + (info.FamilyID == 0xF ? info.ExFamilyID : 0);
|
||||
for (int i = 0; i < CPUCores; ++i) {
|
||||
cpu_stream << "processor : " << i << std::endl;
|
||||
cpu_stream << "processor : " << i << std::endl; // Logical id
|
||||
cpu_stream << "vendor_id : " << vendorid.Str << std::endl;
|
||||
cpu_stream << "cpu family : " << Family << std::endl;
|
||||
cpu_stream << "model : " << (info.Model + (info.FamilyID >= 6 ? (info.ExModelID << 4) : 0)) << std::endl;
|
||||
@@ -88,10 +88,10 @@ namespace FEX::EmulatedFile {
|
||||
cpu_stream << "microcode : 0x0" << std::endl;
|
||||
cpu_stream << "cpu MHz : 3000" << std::endl;
|
||||
cpu_stream << "cache size : 512 KB" << std::endl;
|
||||
cpu_stream << "physical id : " << i << std::endl;
|
||||
cpu_stream << "siblings : " << CPUCores << std::endl;
|
||||
cpu_stream << "core id : " << i << std::endl;
|
||||
cpu_stream << "cpu cores : " << CPUCores << std::endl;
|
||||
cpu_stream << "physical id : 0" << std::endl; // Socket id (always 0 for a single socket system)
|
||||
cpu_stream << "siblings : " << CPUCores << std::endl; // Number of logical cores
|
||||
cpu_stream << "core id : " << i << std::endl; // Physical id
|
||||
cpu_stream << "cpu cores : " << CPUCores << std::endl; // Number of physical cores
|
||||
cpu_stream << "apicid : " << i << std::endl;
|
||||
cpu_stream << "initial apicid : " << i << std::endl;
|
||||
cpu_stream << "fpu : " << (res_1.edx & (1 << 0) ? "yes" : "no") << std::endl;
|
||||
|
||||
@@ -95,6 +95,20 @@ uint64_t FileManager::FAccessat(int dirfd, const char *pathname, int mode) {
|
||||
return ::syscall(SYS_faccessat, dirfd, pathname, mode);
|
||||
}
|
||||
|
||||
uint64_t FileManager::FAccessat2(int dirfd, const char *pathname, int mode, int flags) {
|
||||
#ifndef SYS_faccessat2
|
||||
const uint32_t SYS_faccessat2 = 439;
|
||||
#endif
|
||||
auto Path = GetEmulatedPath(pathname);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::syscall(SYS_faccessat2, dirfd, Path.c_str(), mode, flags);
|
||||
if (Result != -1)
|
||||
return Result;
|
||||
}
|
||||
|
||||
return ::syscall(SYS_faccessat2, dirfd, pathname, mode, flags);
|
||||
}
|
||||
|
||||
uint64_t FileManager::Readlink(const char *pathname, char *buf, size_t bufsiz) {
|
||||
if (strcmp(pathname, "/proc/self/exe") == 0 || strcmp(pathname, PidSelfPath.c_str()) == 0) {
|
||||
auto App = Filename();
|
||||
|
||||
@@ -28,6 +28,7 @@ public:
|
||||
uint64_t Lstat(const char *path, void *buf);
|
||||
uint64_t Access(const char *pathname, int mode);
|
||||
uint64_t FAccessat(int dirfd, const char *pathname, int mode);
|
||||
uint64_t FAccessat2(int dirfd, const char *pathname, int mode, int flags);
|
||||
uint64_t Readlink(const char *pathname, char *buf, size_t bufsiz);
|
||||
uint64_t Chmod(const char *pathname, mode_t mode);
|
||||
uint64_t Readlinkat(int dirfd, const char *pathname, char *buf, size_t bufsiz);
|
||||
|
||||
@@ -64,7 +64,7 @@ uint64_t SyscallHandler::HandleBRK(FEXCore::Core::InternalThreadState *Thread, v
|
||||
|
||||
uint64_t NewBRK = (uint64_t)mmap((void*)(DataSpace + DataSpaceMaxSize), AllocateNewSize, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
if (NewBRK != (DataSpace + DataSpaceMaxSize)) {
|
||||
if (NewBRK != ~0ULL && NewBRK != (DataSpace + DataSpaceMaxSize)) {
|
||||
// Couldn't allocate that the region we wanted
|
||||
// Can happen if MAP_FIXED_NOREPLACE isn't understood by the kernel
|
||||
munmap(reinterpret_cast<void*>(NewBRK), AllocateNewSize);
|
||||
@@ -164,6 +164,15 @@ void SyscallHandler::Strace(FEXCore::HLE::SyscallArguments *Args, uint64_t Ret)
|
||||
}
|
||||
#endif
|
||||
|
||||
uint64_t UnimplementedSyscall(FEXCore::Core::InternalThreadState *Thread, uint64_t SyscallNumber) {
|
||||
ERROR_AND_DIE("Unhandled system call: %d", SyscallNumber);
|
||||
return -ENOSYS;
|
||||
}
|
||||
|
||||
uint64_t UnimplementedSyscallSafe(FEXCore::Core::InternalThreadState *Thread, uint64_t SyscallNumber) {
|
||||
return -ENOSYS;
|
||||
}
|
||||
|
||||
FEX::HLE::SyscallHandler *CreateHandler(FEXCore::Context::OperatingMode Mode,
|
||||
FEXCore::Context::Context *ctx,
|
||||
FEX::HLE::SignalDelegator *_SignalDelegation,
|
||||
|
||||
@@ -24,8 +24,9 @@ struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEX::HLE {
|
||||
class SyscallHandler;
|
||||
void RegisterEpoll();
|
||||
void RegisterFD();
|
||||
void RegisterFD(FEX::HLE::SyscallHandler *const Handler);
|
||||
void RegisterFS();
|
||||
void RegisterInfo();
|
||||
void RegisterIO();
|
||||
@@ -44,6 +45,9 @@ namespace FEX::HLE {
|
||||
void RegisterNotImplemented();
|
||||
void RegisterStubs();
|
||||
|
||||
uint64_t UnimplementedSyscall(FEXCore::Core::InternalThreadState *Thread, uint64_t SyscallNumber);
|
||||
uint64_t UnimplementedSyscallSafe(FEXCore::Core::InternalThreadState *Thread, uint64_t SyscallNumber);
|
||||
|
||||
class SyscallHandler : public FEXCore::HLE::SyscallHandler {
|
||||
public:
|
||||
SyscallHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *_SignalDelegation);
|
||||
|
||||
@@ -36,7 +36,7 @@ namespace FEX::HLE {
|
||||
return flags;
|
||||
}
|
||||
|
||||
void RegisterFD() {
|
||||
void RegisterFD(FEX::HLE::SyscallHandler *const Handler) {
|
||||
REGISTER_SYSCALL_IMPL(read, [](FEXCore::Core::InternalThreadState *Thread, int fd, void *buf, size_t count) -> uint64_t {
|
||||
uint64_t Result = ::read(fd, buf, count);
|
||||
SYSCALL_ERRNO();
|
||||
@@ -228,6 +228,17 @@ namespace FEX::HLE {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
if (Handler->GetHostKernelVersion() >= FEX::HLE::SyscallHandler::KernelVersion(5, 8, 0)) {
|
||||
// Only exists on kernel 5.8+
|
||||
REGISTER_SYSCALL_IMPL(faccessat2, [](FEXCore::Core::InternalThreadState *Thread, int dirfd, const char *pathname, int mode, int flags) -> uint64_t {
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.FAccessat2(dirfd, pathname, mode, flags);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
}
|
||||
else {
|
||||
REGISTER_SYSCALL_IMPL(faccessat2, UnimplementedSyscallSafe);
|
||||
}
|
||||
|
||||
REGISTER_SYSCALL_IMPL(splice, [](FEXCore::Core::InternalThreadState *Thread, int fd_in, loff_t *off_in, int fd_out, loff_t *off_out, size_t len, unsigned int flags) -> uint64_t {
|
||||
uint64_t Result = ::splice(fd_in, off_in, fd_out, off_out, len, flags);
|
||||
SYSCALL_ERRNO();
|
||||
|
||||
@@ -2,6 +2,8 @@
|
||||
#include "Tests/LinuxSyscalls/x64/Syscalls.h"
|
||||
#include "Tests/LinuxSyscalls/x32/Syscalls.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstring>
|
||||
#include <linux/kcmp.h>
|
||||
#include <linux/seccomp.h>
|
||||
@@ -19,6 +21,25 @@
|
||||
|
||||
namespace FEX::HLE {
|
||||
void RegisterInfo() {
|
||||
REGISTER_SYSCALL_IMPL(uname, [](FEXCore::Core::InternalThreadState *Thread, struct utsname *buf) -> uint64_t {
|
||||
struct utsname Local{};
|
||||
if (::uname(&Local) == 0) {
|
||||
memcpy(buf->nodename, Local.nodename, sizeof(Local.nodename));
|
||||
static_assert(sizeof(Local.nodename) <= sizeof(buf->nodename));
|
||||
}
|
||||
else {
|
||||
strcpy(buf->nodename, "FEXCore");
|
||||
LogMan::Msg::E("Couldn't determine host nodename. Defaulting to '%s'", buf->nodename);
|
||||
}
|
||||
strcpy(buf->sysname, "Linux");
|
||||
strcpy(buf->release, "5.0.0");
|
||||
strcpy(buf->version, "#" FEXCORE_VERSION);
|
||||
static_assert(sizeof("#" FEXCORE_VERSION) <= sizeof(buf->version), "FEXCORE_VERSION define became too large!");
|
||||
// Tell the guest that we are a 64bit kernel
|
||||
strcpy(buf->machine, "x86_64");
|
||||
return 0;
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL(getrlimit, [](FEXCore::Core::InternalThreadState *Thread, int resource, struct rlimit *rlim) -> uint64_t {
|
||||
uint64_t Result = ::getrlimit(resource, rlim);
|
||||
SYSCALL_ERRNO();
|
||||
|
||||
@@ -530,6 +530,35 @@ namespace FEX::HLE::x32 {
|
||||
writefds ? &Host_writefds : nullptr,
|
||||
exceptfds ? &Host_exceptfds : nullptr,
|
||||
timeout ? &tp64 : nullptr);
|
||||
if (readfds) {
|
||||
for (int i = 0; i < nfds; ++i) {
|
||||
if (FD_ISSET(i, &Host_readfds)) {
|
||||
readfds[i/32] |= 1 << (i & 31);
|
||||
} else {
|
||||
readfds[i/32] &= ~(1 << (i & 31));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (writefds) {
|
||||
for (int i = 0; i < nfds; ++i) {
|
||||
if (FD_ISSET(i, &Host_writefds)) {
|
||||
writefds[i/32] |= 1 << (i & 31);
|
||||
} else {
|
||||
writefds[i/32] &= ~(1 << (i & 31));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (exceptfds) {
|
||||
for (int i = 0; i < nfds; ++i) {
|
||||
if (FD_ISSET(i, &Host_exceptfds)) {
|
||||
exceptfds[i/32] |= 1 << (i & 31);
|
||||
} else {
|
||||
exceptfds[i/32] &= ~(1 << (i & 31));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (timeout) {
|
||||
*timeout = tp64;
|
||||
@@ -609,6 +638,36 @@ namespace FEX::HLE::x32 {
|
||||
timeout ? &tp64 : nullptr,
|
||||
&HostSet);
|
||||
|
||||
if (readfds) {
|
||||
for (int i = 0; i < nfds; ++i) {
|
||||
if (FD_ISSET(i, &Host_readfds)) {
|
||||
readfds[i/32] |= 1 << (i & 31);
|
||||
} else {
|
||||
readfds[i/32] &= ~(1 << (i & 31));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (writefds) {
|
||||
for (int i = 0; i < nfds; ++i) {
|
||||
if (FD_ISSET(i, &Host_writefds)) {
|
||||
writefds[i/32] |= 1 << (i & 31);
|
||||
} else {
|
||||
writefds[i/32] &= ~(1 << (i & 31));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (exceptfds) {
|
||||
for (int i = 0; i < nfds; ++i) {
|
||||
if (FD_ISSET(i, &Host_exceptfds)) {
|
||||
exceptfds[i/32] |= 1 << (i & 31);
|
||||
} else {
|
||||
exceptfds[i/32] &= ~(1 << (i & 31));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (timeout) {
|
||||
*timeout = tp64;
|
||||
}
|
||||
@@ -703,6 +762,36 @@ namespace FEX::HLE::x32 {
|
||||
timeout,
|
||||
&HostSet);
|
||||
|
||||
if (readfds) {
|
||||
for (int i = 0; i < nfds; ++i) {
|
||||
if (FD_ISSET(i, &Host_readfds)) {
|
||||
readfds[i/32] |= 1 << (i & 31);
|
||||
} else {
|
||||
readfds[i/32] &= ~(1 << (i & 31));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (writefds) {
|
||||
for (int i = 0; i < nfds; ++i) {
|
||||
if (FD_ISSET(i, &Host_writefds)) {
|
||||
writefds[i/32] |= 1 << (i & 31);
|
||||
} else {
|
||||
writefds[i/32] &= ~(1 << (i & 31));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (exceptfds) {
|
||||
for (int i = 0; i < nfds; ++i) {
|
||||
if (FD_ISSET(i, &Host_exceptfds)) {
|
||||
exceptfds[i/32] |= 1 << (i & 31);
|
||||
} else {
|
||||
exceptfds[i/32] &= ~(1 << (i & 31));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
}
|
||||
|
||||
@@ -29,16 +29,6 @@ namespace FEX::HLE::x32 {
|
||||
static_assert(sizeof(sysinfo32) == 64, "Needs to be 64bytes");
|
||||
|
||||
void RegisterInfo() {
|
||||
REGISTER_SYSCALL_IMPL_X32(uname, [](FEXCore::Core::InternalThreadState *Thread, struct utsname *buf) -> uint64_t {
|
||||
strcpy(buf->sysname, "Linux");
|
||||
strcpy(buf->nodename, "FEXCore");
|
||||
strcpy(buf->release, "5.0.0");
|
||||
strcpy(buf->version, "#" FEXCORE_VERSION);
|
||||
// Tell the guest that we are a 64bit kernel
|
||||
strcpy(buf->machine, "x86_64");
|
||||
return 0;
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(ugetrlimit, [](FEXCore::Core::InternalThreadState *Thread, int resource, rlimit32 *rlim) -> uint64_t {
|
||||
struct rlimit rlim64{};
|
||||
uint64_t Result = ::getrlimit(resource, &rlim64);
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
#include "Tests/LinuxSyscalls/Syscalls.h"
|
||||
#include "Tests/LinuxSyscalls/x32/Syscalls.h"
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <bitset>
|
||||
#include <map>
|
||||
@@ -7,27 +8,65 @@
|
||||
#include <stdint.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/ipc.h>
|
||||
#include <unistd.h>
|
||||
#include <filesystem>
|
||||
|
||||
static std::string get_fdpath(int fd)
|
||||
{
|
||||
std::error_code ec;
|
||||
return std::filesystem::canonical(std::filesystem::path("/proc/self/fd") / std::to_string(fd), ec).string();
|
||||
}
|
||||
|
||||
namespace FEX::HLE::x32 {
|
||||
|
||||
void RegisterMemory() {
|
||||
REGISTER_SYSCALL_IMPL_X32(mmap, [](FEXCore::Core::InternalThreadState *Thread, uint32_t addr, uint32_t length, int prot, int flags, int fd, int32_t offset) -> uint64_t {
|
||||
return (uint64_t)static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
auto Result = (uint64_t)static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
mmap(reinterpret_cast<void*>(addr), length, prot,flags, fd, offset);
|
||||
|
||||
if (Result != -1) {
|
||||
if (!(flags & MAP_ANONYMOUS)) {
|
||||
auto filename = get_fdpath(fd);
|
||||
|
||||
FEXCore::Context::AddNamedRegion(Thread->CTX, Result, length, offset, filename);
|
||||
}
|
||||
FEXCore::Context::FlushCodeRange(Thread, (uintptr_t)Result, length);
|
||||
}
|
||||
|
||||
return Result;
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(mmap2, [](FEXCore::Core::InternalThreadState *Thread, uint32_t addr, uint32_t length, int prot, int flags, int fd, uint32_t pgoffset) -> uint64_t {
|
||||
return (uint64_t)static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
auto Result = (uint64_t)static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
mmap(reinterpret_cast<void*>(addr), length, prot,flags, fd, (uint64_t)pgoffset * 0x1000);
|
||||
|
||||
if (Result != -1) {
|
||||
if (!(flags & MAP_ANONYMOUS)) {
|
||||
auto filename = get_fdpath(fd);
|
||||
|
||||
FEXCore::Context::AddNamedRegion(Thread->CTX, Result, length, pgoffset * 0x1000, filename);
|
||||
}
|
||||
FEXCore::Context::FlushCodeRange(Thread, (uintptr_t)Result, length);
|
||||
}
|
||||
|
||||
return Result;
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(munmap, [](FEXCore::Core::InternalThreadState *Thread, void *addr, size_t length) -> uint64_t {
|
||||
return static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
auto Result = static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
munmap(addr, length);
|
||||
if (Result != -1) {
|
||||
FEXCore::Context::RemoveNamedRegion(Thread->CTX, (uintptr_t)addr, length);
|
||||
FEXCore::Context::FlushCodeRange(Thread, (uintptr_t)addr, length);
|
||||
}
|
||||
return Result;
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(mprotect, [](FEXCore::Core::InternalThreadState *Thread, void *addr, uint32_t len, int prot) -> uint64_t {
|
||||
uint64_t Result = ::mprotect(addr, len, prot);
|
||||
if (Result != -1 && prot & PROT_EXEC) {
|
||||
FEXCore::Context::FlushCodeRange(Thread, (uintptr_t)addr, len);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
|
||||
@@ -521,12 +521,6 @@ uint64_t MemAllocator::shmdt(const void* shmaddr) {
|
||||
});
|
||||
}
|
||||
|
||||
uint32_t Unimplemented(FEXCore::Core::InternalThreadState *Thread, uint64_t SyscallNumber) {
|
||||
auto name = GetSyscallName(SyscallNumber);
|
||||
ERROR_AND_DIE("Unhandled system call: %d, %s", SyscallNumber, name);
|
||||
return -ENOSYS;
|
||||
}
|
||||
|
||||
x32SyscallHandler::x32SyscallHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *_SignalDelegation)
|
||||
: SyscallHandler {ctx, _SignalDelegation} {
|
||||
AllocHandler = std::make_unique<MemAllocator>();
|
||||
@@ -548,11 +542,11 @@ uint64_t MemAllocator::shmdt(const void* shmaddr) {
|
||||
// Clear all definitions
|
||||
for (auto &Def : Definitions) {
|
||||
Def.NumArgs = 255;
|
||||
Def.Ptr = cvt(&Unimplemented);
|
||||
Def.Ptr = cvt(&UnimplementedSyscall);
|
||||
}
|
||||
|
||||
FEX::HLE::RegisterEpoll();
|
||||
FEX::HLE::RegisterFD();
|
||||
FEX::HLE::RegisterFD(this);
|
||||
FEX::HLE::RegisterFS();
|
||||
FEX::HLE::RegisterInfo();
|
||||
FEX::HLE::RegisterIO();
|
||||
@@ -590,7 +584,7 @@ uint64_t MemAllocator::shmdt(const void* shmaddr) {
|
||||
auto SyscallNumber = Syscall.SyscallNumber;
|
||||
auto Name = GetSyscallName(SyscallNumber);
|
||||
auto &Def = Definitions.at(SyscallNumber);
|
||||
LogMan::Throw::A(Def.Ptr == cvt(&Unimplemented), "Oops overwriting sysall problem, %d, %s", SyscallNumber, Name);
|
||||
LogMan::Throw::A(Def.Ptr == cvt(&UnimplementedSyscall), "Oops overwriting sysall problem, %d, %s", SyscallNumber, Name);
|
||||
Def.Ptr = Syscall.SyscallHandler;
|
||||
Def.NumArgs = Syscall.ArgumentCount;
|
||||
#ifdef DEBUG_STRACE
|
||||
@@ -600,7 +594,7 @@ uint64_t MemAllocator::shmdt(const void* shmaddr) {
|
||||
|
||||
#if PRINT_MISSING_SYSCALLS
|
||||
for (auto &Syscall: SyscallNames) {
|
||||
if (Definitions[Syscall.first].Ptr == cvt(&Unimplemented)) {
|
||||
if (Definitions[Syscall.first].Ptr == cvt(&UnimplementedSyscall)) {
|
||||
LogMan::Msg::D("Unimplemented syscall: %d: %s", Syscall.first, Syscall.second);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -425,6 +425,12 @@ enum Syscalls {
|
||||
SYSCALL_x86_fspick = 433,
|
||||
SYSCALL_x86_pidfd_open = 434,
|
||||
SYSCALL_x86_clone3 = 435,
|
||||
|
||||
SYSCALL_x86_close_range = 436,
|
||||
SYSCALL_x86_openat2 = 437,
|
||||
SYSCALL_x86_pidfd_getfd = 438,
|
||||
SYSCALL_x86_faccessat2 = 439,
|
||||
SYSCALL_x86_process_madvise = 440,
|
||||
SYSCALL_x86_epoll_pwait2 = 441,
|
||||
SYSCALL_x86_mount_setattr = 442,
|
||||
SYSCALL_MAX = 512,
|
||||
};
|
||||
@@ -9,15 +9,6 @@
|
||||
|
||||
namespace FEX::HLE::x64 {
|
||||
void RegisterInfo() {
|
||||
REGISTER_SYSCALL_IMPL_X64(uname, [](FEXCore::Core::InternalThreadState *Thread, struct utsname *buf) -> uint64_t {
|
||||
strcpy(buf->sysname, "Linux");
|
||||
strcpy(buf->nodename, "FEXCore");
|
||||
strcpy(buf->release, "5.0.0");
|
||||
strcpy(buf->version, "#" FEXCORE_VERSION);
|
||||
strcpy(buf->machine, "x86_64");
|
||||
return 0;
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64(sysinfo, [](FEXCore::Core::InternalThreadState *Thread, struct sysinfo *info) -> uint64_t {
|
||||
uint64_t Result = ::sysinfo(info);
|
||||
SYSCALL_ERRNO();
|
||||
|
||||
@@ -1,19 +1,46 @@
|
||||
#include "Tests/LinuxSyscalls/Syscalls.h"
|
||||
#include "Tests/LinuxSyscalls/x64/Syscalls.h"
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <sys/mman.h>
|
||||
#include <sys/shm.h>
|
||||
#include <map>
|
||||
#include <unistd.h>
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <fstream>
|
||||
#include <filesystem>
|
||||
|
||||
static std::string get_fdpath(int fd)
|
||||
{
|
||||
std::error_code ec;
|
||||
return std::filesystem::canonical(std::filesystem::path("/proc/self/fd") / std::to_string(fd), ec).string();
|
||||
}
|
||||
|
||||
namespace FEX::HLE::x64 {
|
||||
void RegisterMemory() {
|
||||
REGISTER_SYSCALL_IMPL_X64(munmap, [](FEXCore::Core::InternalThreadState *Thread, void *addr, size_t length) -> uint64_t {
|
||||
uint64_t Result = ::munmap(addr, length);
|
||||
if (Result != -1) {
|
||||
FEXCore::Context::RemoveNamedRegion(Thread->CTX, (uintptr_t)addr, length);
|
||||
FEXCore::Context::FlushCodeRange(Thread, (uintptr_t)addr, length);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64(mmap, [](FEXCore::Core::InternalThreadState *Thread, void *addr, size_t length, int prot, int flags, int fd, off_t offset) -> uint64_t {
|
||||
static FEXCore::Config::Value<bool> AOTIRLoad(FEXCore::Config::CONFIG_AOTIR_LOAD, false);
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(::mmap(addr, length, prot, flags, fd, offset));
|
||||
if (Result != -1) {
|
||||
if (!(flags & MAP_ANONYMOUS)) {
|
||||
auto filename = get_fdpath(fd);
|
||||
|
||||
FEXCore::Context::AddNamedRegion(Thread->CTX, Result, length, offset, filename);
|
||||
}
|
||||
FEXCore::Context::FlushCodeRange(Thread, (uintptr_t)Result, length);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
@@ -24,6 +51,9 @@ namespace FEX::HLE::x64 {
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64(mprotect, [](FEXCore::Core::InternalThreadState *Thread, void *addr, size_t len, int prot) -> uint64_t {
|
||||
uint64_t Result = ::mprotect(addr, len, prot);
|
||||
if (Result != -1 && prot & PROT_EXEC) {
|
||||
FEXCore::Context::FlushCodeRange(Thread, (uintptr_t)addr, len);
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
|
||||
@@ -68,15 +68,6 @@ namespace FEX::HLE::x64 {
|
||||
void RegisterSyscallHandlers();
|
||||
};
|
||||
|
||||
uint64_t Unimplemented(FEXCore::Core::InternalThreadState *Thread, uint64_t SyscallNumber) {
|
||||
|
||||
auto name = GetSyscallName(SyscallNumber);
|
||||
|
||||
ERROR_AND_DIE("Unhandled system call: %d, %s", SyscallNumber, name);
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
x64SyscallHandler::x64SyscallHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *_SignalDelegation)
|
||||
: SyscallHandler {ctx, _SignalDelegation} {
|
||||
OSABI = FEXCore::HLE::SyscallOSABI::OS_LINUX64;
|
||||
@@ -98,11 +89,11 @@ namespace FEX::HLE::x64 {
|
||||
// Clear all definitions
|
||||
for (auto &Def : Definitions) {
|
||||
Def.NumArgs = 255;
|
||||
Def.Ptr = cvt(&Unimplemented);
|
||||
Def.Ptr = cvt(&UnimplementedSyscall);
|
||||
}
|
||||
|
||||
FEX::HLE::RegisterEpoll();
|
||||
FEX::HLE::RegisterFD();
|
||||
FEX::HLE::RegisterFD(this);
|
||||
FEX::HLE::RegisterFS();
|
||||
FEX::HLE::RegisterInfo();
|
||||
FEX::HLE::RegisterIO();
|
||||
@@ -141,7 +132,7 @@ namespace FEX::HLE::x64 {
|
||||
auto SyscallNumber = Syscall.SyscallNumber;
|
||||
auto Name = GetSyscallName(SyscallNumber);
|
||||
auto &Def = Definitions.at(SyscallNumber);
|
||||
LogMan::Throw::A(Def.Ptr == cvt(&Unimplemented), "Oops overwriting sysall problem, %d, %s", SyscallNumber, Name);
|
||||
LogMan::Throw::A(Def.Ptr == cvt(&UnimplementedSyscall), "Oops overwriting sysall problem, %d, %s", SyscallNumber, Name);
|
||||
Def.Ptr = Syscall.SyscallHandler;
|
||||
Def.NumArgs = Syscall.ArgumentCount;
|
||||
#ifdef DEBUG_STRACE
|
||||
@@ -151,7 +142,7 @@ namespace FEX::HLE::x64 {
|
||||
|
||||
#if PRINT_MISSING_SYSCALLS
|
||||
for (auto &Syscall: SyscallNames) {
|
||||
if (Definitions[Syscall.first].Ptr == cvt(&Unimplemented)) {
|
||||
if (Definitions[Syscall.first].Ptr == cvt(&UnimplementedSyscall)) {
|
||||
LogMan::Msg::D("Unimplemented syscall: %s", Syscall.second);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -335,6 +335,24 @@ enum Syscalls {
|
||||
SYSCALL_x64_statx = 332,
|
||||
SYSCALL_x64_io_pgetevents = 333,
|
||||
SYSCALL_x64_rseq = 334,
|
||||
|
||||
SYSCALL_x64_pidfd_send_signal = 424,
|
||||
SYSCALL_x64_io_uring_setup = 425,
|
||||
SYSCALL_x64_io_uring_enter = 426,
|
||||
SYSCALL_x64_io_uring_register = 427,
|
||||
SYSCALL_x64_open_tree = 428,
|
||||
SYSCALL_x64_move_mount = 429,
|
||||
SYSCALL_x64_fsopen = 430,
|
||||
SYSCALL_x64_fsconfig = 431,
|
||||
SYSCALL_x64_fsmount = 432,
|
||||
SYSCALL_x64_fspick = 433,
|
||||
SYSCALL_x64_pidfd_open = 434,
|
||||
SYSCALL_x64_clone3 = 435,
|
||||
SYSCALL_x64_close_range = 436,
|
||||
SYSCALL_x64_openat2 = 437,
|
||||
SYSCALL_x64_pidfd_getfd = 438,
|
||||
SYSCALL_x64_faccessat2 = 439,
|
||||
SYSCALL_x64_process_madvise = 440,
|
||||
SYSCALL_x64_epoll_pwait2 = 441,
|
||||
SYSCALL_x64_mount_setattr = 442,
|
||||
SYSCALL_MAX = 512,
|
||||
};
|
||||
@@ -71,7 +71,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Config::Value<std::string> OutputLog{FEXCore::Config::CONFIG_OUTPUTLOG, "stderr"};
|
||||
FEXCore::Config::Value<std::string> DumpIR{FEXCore::Config::CONFIG_DUMPIR, "no"};
|
||||
FEXCore::Config::Value<bool> TSOEnabledConfig{FEXCore::Config::CONFIG_TSO_ENABLED, true};
|
||||
FEXCore::Config::Value<bool> SMCChecksConfig{FEXCore::Config::CONFIG_SMC_CHECKS, false};
|
||||
FEXCore::Config::Value<uint8_t> SMCChecksConfig{FEXCore::Config::CONFIG_SMC_CHECKS, FEXCore::Config::CONFIG_SMC_MMAN};
|
||||
FEXCore::Config::Value<bool> ABILocalFlags{FEXCore::Config::CONFIG_ABI_LOCAL_FLAGS, false};
|
||||
FEXCore::Config::Value<bool> AbiNoPF{FEXCore::Config::CONFIG_ABI_NO_PF, false};
|
||||
|
||||
|
||||
@@ -663,19 +663,18 @@ namespace IR {
|
||||
FEXCore::Core::ThreadState *State = FEXCore::Context::Debug::GetThreadState(FEX::DebuggerState::GetContext());
|
||||
FEXCore::Core::InternalThreadState *TS = reinterpret_cast<FEXCore::Core::InternalThreadState*>(State);
|
||||
|
||||
auto &IRList = TS->IRLists;
|
||||
auto Local = TS->LocalIRCache;
|
||||
auto &DebugData = TS->DebugData;
|
||||
|
||||
for (auto &IR : IRList) {
|
||||
for (auto &LocalEntry : TS->LocalIRCache) {
|
||||
std::ostringstream out;
|
||||
out << "0x" << std::hex << IR.first;
|
||||
auto Data = DebugData.find(IR.first);
|
||||
out << "0x" << std::hex << LocalEntry.first;
|
||||
IRDebugData DebugData;
|
||||
DebugData.Debug = &Data->second;
|
||||
DebugData.Debug = LocalEntry.second.DebugData.get();
|
||||
DebugData.RIP = IR.first;
|
||||
DebugData.RIPString = out.str();
|
||||
DebugData.GuestCodeSize = std::to_string(Data->second.GuestCodeSize);
|
||||
DebugData.GuestInstructionCount = std::to_string(Data->second.GuestInstructionCount);
|
||||
DebugData.GuestCodeSize = std::to_string(DebugData.Debug->GuestCodeSize);
|
||||
DebugData.GuestInstructionCount = std::to_string(DebugData.Debug->GuestInstructionCount);
|
||||
IRListTexts.emplace_back(DebugData);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -301,11 +301,27 @@ namespace {
|
||||
ConfigChanged = true;
|
||||
}
|
||||
|
||||
ImGui::Text("SMC Checks: ");
|
||||
int SMCChecks = FEXCore::Config::CONFIG_SMC_MMAN;
|
||||
|
||||
Value = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_SMC_CHECKS);
|
||||
bool SMCChecks = Value.has_value() && **Value == "1";
|
||||
if (ImGui::Checkbox("SMC Checks", &SMCChecks)) {
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_SMC_CHECKS, SMCChecks ? "1" : "0");
|
||||
ConfigChanged = true;
|
||||
if (Value.has_value()) {
|
||||
if (**Value == "0") {
|
||||
SMCChecks = FEXCore::Config::CONFIG_SMC_NONE;
|
||||
} else if (**Value == "1") {
|
||||
SMCChecks = FEXCore::Config::CONFIG_SMC_MMAN;
|
||||
} else if (**Value == "2") {
|
||||
SMCChecks = FEXCore::Config::CONFIG_SMC_FULL;
|
||||
}
|
||||
}
|
||||
|
||||
bool SMCChanged = false;
|
||||
SMCChanged |= ImGui::RadioButton("None", &SMCChecks, FEXCore::Config::CONFIG_SMC_NONE); ImGui::SameLine();
|
||||
SMCChanged |= ImGui::RadioButton("MMan", &SMCChecks, FEXCore::Config::CONFIG_SMC_MMAN); ImGui::SameLine();
|
||||
SMCChanged |= ImGui::RadioButton("Full", &SMCChecks, FEXCore::Config::CONFIG_SMC_FULL);
|
||||
|
||||
if (SMCChanged) {
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_SMC_CHECKS, std::to_string(SMCChecks));
|
||||
}
|
||||
|
||||
Value = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_ABI_LOCAL_FLAGS);
|
||||
|
||||
@@ -1,11 +1,18 @@
|
||||
#!/usr/bin/python3
|
||||
import sys
|
||||
import re
|
||||
from hashlib import sha256
|
||||
|
||||
Libs = { }
|
||||
CurrentLib = None
|
||||
CurrentFunction = None
|
||||
|
||||
def hash_lib_fn_asm(lib, fn):
|
||||
return "0x" + ", 0x".join(re.findall('..', sha256((lib + ":" + fn).encode('utf-8')).hexdigest()))
|
||||
|
||||
def hash_lib_fn_c(lib, fn):
|
||||
return "\\x" + "\\x".join(re.findall('..', sha256((lib + ":" + fn).encode('utf-8')).hexdigest()))
|
||||
|
||||
def lib(name):
|
||||
global Libs
|
||||
global CurrentLib
|
||||
@@ -127,8 +134,7 @@ def GenerateThunk_args_assignment(args):
|
||||
return "".join(rv)
|
||||
|
||||
def GenerateFunctionThunk(lib, function):
|
||||
print("MAKE_THUNK(" + lib["name"] + ", " + function["name"] + ")")
|
||||
print("")
|
||||
print("MAKE_THUNK(" + lib["name"] + ", " + function["name"] + ", \"" + hash_lib_fn_asm(lib["name"], function["name"]) + "\")")
|
||||
|
||||
|
||||
###
|
||||
@@ -232,7 +238,7 @@ def GenerateLdr(libs):
|
||||
|
||||
# Used to initialize host thunk list
|
||||
def GenerateTabFunctionUnpack(lib, function):
|
||||
print("{\"" + lib["name"] + ":" + function["name"] + "\", &fexfn_unpack_" + lib["name"] + "_" + function["name"] + "},")
|
||||
print("{(uint8_t*)\"" + hash_lib_fn_c(lib["name"], function["name"]) + "\", &fexfn_unpack_" + lib["name"] + "_" + function["name"] + "}, //", lib["name"] + ":" + function["name"])
|
||||
|
||||
# Symtab, used for glxGetProc
|
||||
def GenerateTabFunctionPack(lib, function):
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
#pragma once
|
||||
#include <stdint.h>
|
||||
|
||||
#define MAKE_THUNK(lib, name) static __attribute__((naked)) int fexthunks_##lib##_##name(void *args) { asm(".byte 0xF, 0x3F"); asm(".asciz \"" #lib ":" #name "\""); }
|
||||
#define MAKE_THUNK(lib, name, hash) extern "C" { int fexthunks_##lib##_##name(void *args); } asm("fexthunks_" #lib "_" #name ":\n.byte 0xF, 0x3F\n.byte " hash );
|
||||
|
||||
struct LoadlibArgs {
|
||||
const char *Name;
|
||||
uintptr_t CallbackThunks;
|
||||
};
|
||||
|
||||
#define LOAD_LIB(name) MAKE_THUNK(fex, loadlib) __attribute__((constructor)) static void loadlib() { LoadlibArgs args = { #name, 0 }; fexthunks_fex_loadlib(&args); }
|
||||
#define LOAD_LIB_WITH_CALLBACKS(name) MAKE_THUNK(fex, loadlib) __attribute__((constructor)) static void loadlib() { LoadlibArgs args = { #name, (uintptr_t)&callback_unpacks }; fexthunks_fex_loadlib(&args); }
|
||||
#define LOAD_LIB(name) MAKE_THUNK(fex, loadlib, "0x27, 0x7e, 0xb7, 0x69, 0x5b, 0xe9, 0xab, 0x12, 0x6e, 0xf7, 0x85, 0x9d, 0x4b, 0xc9, 0xa2, 0x44, 0x46, 0xcf, 0xbd, 0xb5, 0x87, 0x43, 0xef, 0x28, 0xa2, 0x65, 0xba, 0xfc, 0x89, 0x0f, 0x77, 0x80") __attribute__((constructor)) static void loadlib() { LoadlibArgs args = { #name, 0 }; fexthunks_fex_loadlib(&args); }
|
||||
#define LOAD_LIB_WITH_CALLBACKS(name) MAKE_THUNK(fex, loadlib, "0x27, 0x7e, 0xb7, 0x69, 0x5b, 0xe9, 0xab, 0x12, 0x6e, 0xf7, 0x85, 0x9d, 0x4b, 0xc9, 0xa2, 0x44, 0x46, 0xcf, 0xbd, 0xb5, 0x87, 0x43, 0xef, 0x28, 0xa2, 0x65, 0xba, 0xfc, 0x89, 0x0f, 0x77, 0x80") __attribute__((constructor)) static void loadlib() { LoadlibArgs args = { #name, (uintptr_t)&callback_unpacks }; fexthunks_fex_loadlib(&args); }
|
||||
@@ -1,7 +1,7 @@
|
||||
#pragma once
|
||||
#include <stdint.h>
|
||||
|
||||
struct ExportEntry { const char* name; void(*fn)(void *); };
|
||||
struct ExportEntry { uint8_t* sha256; void(*fn)(void *); };
|
||||
|
||||
typedef void fex_call_callback_t(uintptr_t callback, void *arg0, void* arg1);
|
||||
|
||||
|
||||
@@ -74,7 +74,7 @@ foreach(ASM_SRC ${ASM_SOURCES})
|
||||
string(REPLACE " " ";" ARGS_LIST ${ARGS})
|
||||
|
||||
if (TEST_NAME MATCHES "SelfModifyingCode")
|
||||
list(APPEND ARGS_LIST "--smc-full-checks")
|
||||
list(APPEND ARGS_LIST "--smc-checks=full")
|
||||
endif()
|
||||
|
||||
add_test(NAME ${TEST_NAME}
|
||||
|
||||
@@ -34,3 +34,32 @@ Test_TwoByte/0F_B0_7.asm
|
||||
Test_Secondary/09_XX_01_7.asm
|
||||
Test_Secondary/09_XX_01_8.asm
|
||||
Test_Secondary/09_XX_01_9.asm
|
||||
|
||||
# Doesn't support unaligned atomic memory ops on armv8.0
|
||||
Test_Primary/Primary_01_Atomic16.asm
|
||||
Test_Primary/Primary_01_Atomic32.asm
|
||||
Test_Primary/Primary_01_Atomic64.asm
|
||||
Test_Primary/Primary_09_Atomic16.asm
|
||||
Test_Primary/Primary_09_Atomic32.asm
|
||||
Test_Primary/Primary_09_Atomic64.asm
|
||||
Test_Primary/Primary_23_Atomic16.asm
|
||||
Test_Primary/Primary_23_Atomic32.asm
|
||||
Test_Primary/Primary_23_Atomic64.asm
|
||||
Test_Primary/Primary_29_Atomic16.asm
|
||||
Test_Primary/Primary_29_Atomic32.asm
|
||||
Test_Primary/Primary_29_Atomic64.asm
|
||||
Test_Primary/Primary_31_Atomic16.asm
|
||||
Test_Primary/Primary_31_Atomic32.asm
|
||||
Test_Primary/Primary_31_Atomic64.asm
|
||||
Test_Primary/Primary_87_Atomic16.asm
|
||||
Test_Primary/Primary_87_Atomic32.asm
|
||||
Test_Primary/Primary_87_Atomic64.asm
|
||||
Test_Primary/Primary_FF_0_Atomic16.asm
|
||||
Test_Primary/Primary_FF_0_Atomic32.asm
|
||||
Test_Primary/Primary_FF_0_Atomic64.asm
|
||||
Test_Primary/Primary_FF_1_Atomic16.asm
|
||||
Test_Primary/Primary_FF_1_Atomic32.asm
|
||||
Test_Primary/Primary_FF_1_Atomic64.asm
|
||||
Test_TwoByte/0F_C0_Atomic16.asm
|
||||
Test_TwoByte/0F_C0_Atomic32.asm
|
||||
Test_TwoByte/0F_C0_Atomic64.asm
|
||||
@@ -0,0 +1,52 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4242434446464848",
|
||||
"RBX": "0x4242434445464748",
|
||||
"RCX": "0x4142434445464748",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x4242434445464748",
|
||||
"RDI": "0x4142434445464748"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 1 byte offset within 4byte boundary
|
||||
lock add word [r15 + 8 * 0 + 1], ax
|
||||
|
||||
; Test 3 byte offset across 4byte boundary
|
||||
lock add word [r15 + 8 * 0 + 3], ax
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock add word [r15 + 8 * 0 + 7], ax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock add word [r15 + 8 * 0 + 15], ax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock add word [r15 + 8 * 0 + 63], ax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,49 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4242434446464748",
|
||||
"RBX": "0x4242434445464748",
|
||||
"RCX": "0x4142434445464748",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x4242434445464748",
|
||||
"RDI": "0x4142434445464748"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 3 byte offset across 4byte boundary
|
||||
lock add dword [r15 + 8 * 0 + 3], eax
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock add dword [r15 + 8 * 0 + 7], eax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock add dword [r15 + 8 * 0 + 15], eax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock add dword [r15 + 8 * 0 + 63], eax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,46 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4242434445464748",
|
||||
"RBX": "0x4242434445464748",
|
||||
"RCX": "0x4142434445464748",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x4242434445464748",
|
||||
"RDI": "0x4142434445464748"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock add qword [r15 + 8 * 0 + 7], rax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock add qword [r15 + 8 * 0 + 15], rax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock add qword [r15 + 8 * 0 + 63], rax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,52 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4142434445464748",
|
||||
"RBX": "0x4142434445464748",
|
||||
"RCX": "0x4142434445464748",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x4142434445464748",
|
||||
"RDI": "0x4142434445464748"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 1 byte offset within 4byte boundary
|
||||
lock or word [r15 + 8 * 0 + 1], ax
|
||||
|
||||
; Test 3 byte offset across 4byte boundary
|
||||
lock or word [r15 + 8 * 0 + 3], ax
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock or word [r15 + 8 * 0 + 7], ax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock or word [r15 + 8 * 0 + 15], ax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock or word [r15 + 8 * 0 + 63], ax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,49 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4142434445464748",
|
||||
"RBX": "0x4142434445464748",
|
||||
"RCX": "0x4142434445464748",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x4142434445464748",
|
||||
"RDI": "0x4142434445464748"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 3 byte offset across 4byte boundary
|
||||
lock or dword [r15 + 8 * 0 + 3], eax
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock or dword [r15 + 8 * 0 + 7], eax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock or dword [r15 + 8 * 0 + 15], eax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock or dword [r15 + 8 * 0 + 63], eax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,46 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4142434445464748",
|
||||
"RBX": "0x4142434445464748",
|
||||
"RCX": "0x4142434445464748",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x4142434445464748",
|
||||
"RDI": "0x4142434445464748"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock or qword [r15 + 8 * 0 + 7], rax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock or qword [r15 + 8 * 0 + 15], rax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock or qword [r15 + 8 * 0 + 63], rax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,52 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x0142430001000148",
|
||||
"RBX": "0x0142434445464700",
|
||||
"RCX": "0x4142434445464700",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x0142434445464748",
|
||||
"RDI": "0x4142434445464700"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 1 byte offset within 4byte boundary
|
||||
lock and word [r15 + 8 * 0 + 1], ax
|
||||
|
||||
; Test 3 byte offset across 4byte boundary
|
||||
lock and word [r15 + 8 * 0 + 3], ax
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock and word [r15 + 8 * 0 + 7], ax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock and word [r15 + 8 * 0 + 15], ax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock and word [r15 + 8 * 0 + 63], ax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,49 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x0100000001464748",
|
||||
"RBX": "0x0142434445000000",
|
||||
"RCX": "0x4142434445000000",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x0142434445464748",
|
||||
"RDI": "0x4142434445000000"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 3 byte offset across 4byte boundary
|
||||
lock and dword [r15 + 8 * 0 + 3], eax
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock and dword [r15 + 8 * 0 + 7], eax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock and dword [r15 + 8 * 0 + 15], eax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock and dword [r15 + 8 * 0 + 63], eax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,46 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x0142434445464748",
|
||||
"RBX": "0x0100000000000000",
|
||||
"RCX": "0x4100000000000000",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x0142434445464748",
|
||||
"RDI": "0x4100000000000000"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock and qword [r15 + 8 * 0 + 7], rax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock and qword [r15 + 8 * 0 + 15], rax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock and qword [r15 + 8 * 0 + 63], rax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,52 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4042434444464648",
|
||||
"RBX": "0x4042434445464748",
|
||||
"RCX": "0x4142434445464748",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x4042434445464748",
|
||||
"RDI": "0x4142434445464748"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 1 byte offset within 4byte boundary
|
||||
lock sub word [r15 + 8 * 0 + 1], ax
|
||||
|
||||
; Test 3 byte offset across 4byte boundary
|
||||
lock sub word [r15 + 8 * 0 + 3], ax
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock sub word [r15 + 8 * 0 + 7], ax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock sub word [r15 + 8 * 0 + 15], ax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock sub word [r15 + 8 * 0 + 63], ax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,49 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4042434444464748",
|
||||
"RBX": "0x4042434445464748",
|
||||
"RCX": "0x4142434445464748",
|
||||
"RDX": "0x4142434445464748",
|
||||
"RSI": "0x4042434445464748",
|
||||
"RDI": "0x4142434445464748"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov r15, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [r15 + 8 * 0], rax
|
||||
mov [r15 + 8 * 1], rax
|
||||
mov [r15 + 8 * 2], rax
|
||||
mov [r15 + 8 * 3], rax
|
||||
mov [r15 + 8 * 4], rax
|
||||
mov [r15 + 8 * 5], rax
|
||||
mov [r15 + 8 * 6], rax
|
||||
mov [r15 + 8 * 7], rax
|
||||
mov [r15 + 8 * 8], rax
|
||||
mov [r15 + 8 * 9], rax
|
||||
|
||||
mov rax, 1
|
||||
|
||||
; Test 3 byte offset across 4byte boundary
|
||||
lock sub dword [r15 + 8 * 0 + 3], eax
|
||||
|
||||
; Test 7 byte offset across 8byte boundary
|
||||
lock sub dword [r15 + 8 * 0 + 7], eax
|
||||
|
||||
; Test 15 byte offset across 16byte boundary
|
||||
lock sub dword [r15 + 8 * 0 + 15], eax
|
||||
|
||||
; Test 63 byte offset across cacheline boundary
|
||||
lock sub dword [r15 + 8 * 0 + 63], eax
|
||||
|
||||
mov rax, qword [r15 + 8 * 0]
|
||||
mov rbx, qword [r15 + 8 * 1]
|
||||
mov rcx, qword [r15 + 8 * 2]
|
||||
mov rdx, qword [r15 + 8 * 3]
|
||||
mov rsi, qword [r15 + 8 * 7]
|
||||
mov rdi, qword [r15 + 8 * 8]
|
||||
|
||||
hlt
|
||||
Loaded 100 of 136 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user