mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 23:00:17 +02:00
Compare commits
No files matched your search
@@ -76,3 +76,10 @@ jobs:
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gvisor_tests
|
||||
|
||||
- name: gcc target tests 64
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_64
|
||||
|
||||
@@ -26,3 +26,7 @@
|
||||
shallow = true
|
||||
path = External/fex-gvisor-tests-bins
|
||||
url = https://github.com/FEX-Emu/fex-gvisor-tests-bins.git
|
||||
[submodule "External/fex-gcc-target-tests-bins"]
|
||||
shallow = true
|
||||
path = External/fex-gcc-target-tests-bins
|
||||
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
|
||||
+38
-16
@@ -2,6 +2,7 @@ cmake_minimum_required(VERSION 3.14)
|
||||
project(FEX)
|
||||
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
@@ -92,6 +93,9 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
add_definitions(-D_M_ARM_64=1)
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(External/vixl/src/)
|
||||
if(CMAKE_BUILD_TYPE MATCHES DEBUG)
|
||||
add_definitions(-DVIXL_DEBUG=1)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
@@ -143,6 +147,22 @@ if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -march=native")
|
||||
endif()
|
||||
|
||||
if(_M_ARM_64)
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
|
||||
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
|
||||
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=${AARCH64_CPU}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
add_compile_options(-Wall)
|
||||
|
||||
add_subdirectory(External/FEXCore)
|
||||
@@ -156,21 +176,23 @@ if (BUILD_TESTS)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
|
||||
include(ExternalProject)
|
||||
if (BUILD_THUNKS)
|
||||
include(ExternalProject)
|
||||
|
||||
ExternalProject_Add(host-libs
|
||||
PREFIX host-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/HostLibs"
|
||||
BINARY_DIR "Host"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
)
|
||||
ExternalProject_Add(host-libs
|
||||
PREFIX host-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/HostLibs"
|
||||
BINARY_DIR "Host"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
)
|
||||
|
||||
ExternalProject_Add(guest-libs
|
||||
PREFIX guest-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS "-DX86_C_COMPILER:STRING=${X86_C_COMPILER}" "-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
)
|
||||
ExternalProject_Add(guest-libs
|
||||
PREFIX guest-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS "-DX86_C_COMPILER:STRING=${X86_C_COMPILER}" "-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
)
|
||||
endif()
|
||||
+2
-2
@@ -3,7 +3,7 @@ FROM ubuntu:20.04 as builder
|
||||
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
|
||||
libboost-dev clang-10 llvm-10 nasm ninja-build libnuma-dev \
|
||||
clang-10 llvm-10 nasm ninja-build libnuma-dev \
|
||||
libcap-dev libglfw3-dev libepoxy-dev
|
||||
|
||||
COPY . /opt/FEX
|
||||
@@ -21,7 +21,7 @@ RUN ninja
|
||||
FROM ubuntu:20.04
|
||||
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt install -y libboost-dev \
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt install -y \
|
||||
libnuma-dev libcap-dev libglfw3-dev libepoxy-dev
|
||||
|
||||
COPY --from=builder /opt/FEX/build/Bin/* /usr/bin/
|
||||
|
||||
Vendored
+8
@@ -43,6 +43,7 @@ set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
find_package(Git)
|
||||
|
||||
set(GIT_SHORT_HASH "Unknown")
|
||||
set(GIT_DESCRIBE_STRING "FEX-Unknown")
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
@@ -52,6 +53,13 @@ if (GIT_FOUND)
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
endif()
|
||||
|
||||
configure_file(
|
||||
|
||||
+1
-1
@@ -236,7 +236,7 @@ def print_ir_arg_printer(ops, defines):
|
||||
if (SSAArgs != 0):
|
||||
for i in range(0, SSAArgs):
|
||||
LastArg = (SSAArgs - i - 1) == 0 and not HasArgs
|
||||
output_file.write("\tPrintArg(out, IR, Op->Header.Args[%d], RAPass);\n" % i)
|
||||
output_file.write("\tPrintArg(out, IR, Op->Header.Args[%d], RAData);\n" % i)
|
||||
if not (LastArg):
|
||||
output_file.write("\t*out << \", \";\n")
|
||||
|
||||
|
||||
+8
-9
@@ -118,7 +118,7 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
Interface/Config/Config.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/BlockCache.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
Interface/Core/CompileService.cpp
|
||||
Interface/Core/Core.cpp
|
||||
@@ -131,6 +131,7 @@ set (SRCS
|
||||
Interface/Core/X86DebugInfo.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
Interface/Core/Interpreter/InterpreterOps.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
@@ -144,7 +145,8 @@ set (SRCS
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/IR/IR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IRParser.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
Interface/IR/PassManager.cpp
|
||||
Interface/IR/Passes/ConstProp.cpp
|
||||
@@ -155,10 +157,8 @@ set (SRCS
|
||||
Interface/IR/Passes/ValueDominanceValidation.cpp
|
||||
Interface/IR/Passes/PhiValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/DeadFlagStoreElimination.cpp
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/StaticRegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/DeadGPRStoreElimination.cpp
|
||||
Interface/IR/Passes/DeadFPRStoreElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/ELFLoader.cpp
|
||||
@@ -170,7 +170,9 @@ if (_M_X86_64)
|
||||
list(APPEND SRCS Interface/Core/Interpreter/x86_64Dispatcher.cpp)
|
||||
endif()
|
||||
if(_M_ARM_64)
|
||||
list(APPEND SRCS Interface/Core/Interpreter/Arm64Dispatcher.cpp)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/ArchHelpers/Arm64.cpp
|
||||
Interface/Core/Interpreter/Arm64Dispatcher.cpp)
|
||||
endif()
|
||||
|
||||
set (JIT_LIBS )
|
||||
@@ -259,9 +261,6 @@ add_custom_target(IR_INC
|
||||
check_cxx_compiler_flag(-fdiagnostics-color=always GCC_COLOR)
|
||||
check_cxx_compiler_flag(-fcolor-diagnostics CLANG_COLOR)
|
||||
|
||||
set(LINUX_LIBS
|
||||
numa)
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
|
||||
@@ -103,7 +103,9 @@ float32_t
|
||||
if ( ! sig ) exp = 0;
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
packReturn:
|
||||
#endif
|
||||
uiZ = packToF32UI( sign, exp, sig );
|
||||
uiZ:
|
||||
uZ.ui = uiZ;
|
||||
|
||||
@@ -107,7 +107,9 @@ float64_t
|
||||
if ( ! sig ) exp = 0;
|
||||
/*------------------------------------------------------------------------
|
||||
*------------------------------------------------------------------------*/
|
||||
#ifdef SOFTFLOAT_ROUND_ODD
|
||||
packReturn:
|
||||
#endif
|
||||
uiZ = packToF64UI( sign, exp, sig );
|
||||
uiZ:
|
||||
uZ.ui = uiZ;
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <map>
|
||||
|
||||
namespace FEXCore::Config {
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
|
||||
@@ -41,6 +42,9 @@ namespace FEXCore::Config {
|
||||
case FEXCore::Config::CONFIG_ABI_NO_PF:
|
||||
CTX->Config.ABINoPF = Config != 0;
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_VALIDATE_IR_PARSER:
|
||||
CTX->Config.ValidateIRarser = Config != 0;
|
||||
break;
|
||||
default: LogMan::Msg::A("Unknown configuration option");
|
||||
}
|
||||
}
|
||||
@@ -94,6 +98,9 @@ namespace FEXCore::Config {
|
||||
case FEXCore::Config::CONFIG_ABI_NO_PF:
|
||||
return CTX->Config.ABINoPF;
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_VALIDATE_IR_PARSER:
|
||||
return CTX->Config.ValidateIRarser;
|
||||
break;
|
||||
default: LogMan::Msg::A("Unknown configuration option");
|
||||
}
|
||||
|
||||
|
||||
+11
-4
@@ -65,8 +65,14 @@ namespace FEXCore::Context {
|
||||
|
||||
std::string DumpIR;
|
||||
|
||||
// this is for internal use
|
||||
bool ValidateIRarser { false };
|
||||
|
||||
} Config;
|
||||
|
||||
using IntCallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
IntCallbackReturn InterpreterCallbackReturn;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
|
||||
std::mutex ThreadCreationMutex;
|
||||
@@ -136,9 +142,10 @@ namespace FEXCore::Context {
|
||||
FEXCore::Core::ThreadState *GetThreadState();
|
||||
void LoadEntryList();
|
||||
|
||||
std::tuple<void *, FEXCore::Core::DebugData *> CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
std::tuple<FEXCore::IR::IRListView<true> *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t> GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
std::tuple<void *, FEXCore::IR::IRListView<true> *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool> CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
uintptr_t CompileBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
uintptr_t CompileFallbackBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
|
||||
@@ -155,7 +162,7 @@ namespace FEXCore::Context {
|
||||
#endif
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
|
||||
|
||||
private:
|
||||
void WaitForIdleWithTimeout();
|
||||
@@ -163,7 +170,7 @@ namespace FEXCore::Context {
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void NotifyPause();
|
||||
|
||||
uintptr_t AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
|
||||
|
||||
FEXCore::CodeLoader *LocalLoader{};
|
||||
|
||||
|
||||
@@ -0,0 +1,841 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <stdint.h>
|
||||
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
static uint64_t LoadAcquire64(uint64_t Addr) {
|
||||
std::atomic<uint64_t> *Atom = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
return Atom->load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static bool StoreCAS64(uint64_t &Expected, uint64_t Val, uint64_t Addr) {
|
||||
std::atomic<uint64_t> *Atom = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
static uint32_t LoadAcquire32(uint64_t Addr) {
|
||||
std::atomic<uint32_t> *Atom = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
return Atom->load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static bool StoreCAS32(uint32_t &Expected, uint32_t Val, uint64_t Addr) {
|
||||
std::atomic<uint32_t> *Atom = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
static uint8_t LoadAcquire8(uint64_t Addr) {
|
||||
std::atomic<uint8_t> *Atom = reinterpret_cast<std::atomic<uint8_t>*>(Addr);
|
||||
return Atom->load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static bool StoreCAS8(uint8_t &Expected, uint8_t Val, uint64_t Addr) {
|
||||
std::atomic<uint8_t> *Atom = reinterpret_cast<std::atomic<uint8_t>*>(Addr);
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
bool HandleCASPAL(void *_mcontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = reinterpret_cast<mcontext_t*>(_mcontext);
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
|
||||
uint32_t DesiredReg1 = Instr & 0b11111;
|
||||
uint32_t DesiredReg2 = DesiredReg1 + 1;
|
||||
uint32_t ExpectedReg1 = (Instr >> 16) & 0b11111;
|
||||
uint32_t ExpectedReg2 = ExpectedReg1 + 1;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
if (Size == 0) {
|
||||
// 32bit
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
uint32_t DesiredLower = mcontext->regs[DesiredReg1];
|
||||
uint32_t DesiredUpper = mcontext->regs[DesiredReg2];
|
||||
|
||||
uint32_t ExpectedLower = mcontext->regs[ExpectedReg1];
|
||||
uint32_t ExpectedUpper = mcontext->regs[ExpectedReg2];
|
||||
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
// It isn't even guaranteed to work on x86
|
||||
// Intel will do a "split lock" which locks the full bus
|
||||
// AMD will tear instead
|
||||
// Both cross-cacheline and cross 16byte both need dual CAS loops that can tear
|
||||
// ARMv8.4 LSE2 solves all atomic issues except cross-cacheline
|
||||
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
uint64_t Alignment = Addr & 0b111;
|
||||
Addr &= ~0b111ULL;
|
||||
uint64_t AddrUpper = Addr + 8;
|
||||
|
||||
// Crosses a 16byte boundary
|
||||
// Need to do 256bit atomic, but since that doesn't exist we need to do a dual CAS loop
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = DesiredUpper;
|
||||
Desired <<= 32;
|
||||
Desired |= DesiredLower;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = ExpectedUpper;
|
||||
Expected <<= 32;
|
||||
Expected |= ExpectedLower;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
__uint128_t LoadOrderUpper = LoadAcquire64(AddrUpper);
|
||||
LoadOrderUpper <<= 64;
|
||||
__uint128_t TmpActual = LoadOrderUpper | LoadAcquire64(Addr);
|
||||
|
||||
// Set up expected
|
||||
TmpExpected = TmpActual;
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
uint64_t TmpExpectedLower = TmpExpected;
|
||||
uint64_t TmpExpectedUpper = TmpExpected >> 64;
|
||||
|
||||
uint64_t TmpDesiredLower = TmpDesired;
|
||||
uint64_t TmpDesiredUpper = TmpDesired >> 64;
|
||||
|
||||
if (TmpExpected == TmpActual) {
|
||||
if (StoreCAS64(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS64(TmpExpectedLower, TmpDesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
}
|
||||
}
|
||||
|
||||
TmpExpected = TmpExpectedUpper;
|
||||
TmpExpected <<= 64;
|
||||
TmpExpected |= TmpExpectedLower;
|
||||
}
|
||||
else {
|
||||
// Mismatch up front
|
||||
TmpExpected = TmpActual;
|
||||
}
|
||||
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t> *Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = (uint64_t)DesiredUpper << 32 | DesiredLower;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = (uint64_t)ExpectedUpper << 32 | ExpectedLower;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleCASAL(void *_mcontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = reinterpret_cast<mcontext_t*>(_mcontext);
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t DesiredReg = Instr & 0b11111;
|
||||
uint32_t ExpectedReg = (Instr >> 16) & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
// It isn't even guaranteed to work on x86
|
||||
// Intel will do a "split lock" which locks the full bus
|
||||
// AMD will tear instead
|
||||
// Both cross-cacheline and cross 16byte both need dual CAS loops that can tear
|
||||
// ARMv8.4 LSE2 solves all atomic issues except cross-cacheline
|
||||
|
||||
// 8bit can't be unaligned
|
||||
// Only need to handle 16, 32, 64
|
||||
if (Size == 2) {
|
||||
// 16 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) == 15) {
|
||||
// Address crosses over 16byte or 64byte threshold
|
||||
// Need a dual 8bit CAS loop
|
||||
uint64_t AddrUpper = Addr + 1;
|
||||
|
||||
uint16_t Desired = mcontext->regs[DesiredReg];
|
||||
uint16_t Expected = mcontext->regs[ExpectedReg];
|
||||
|
||||
uint8_t DesiredLower = Desired;
|
||||
uint8_t DesiredUpper = Desired >> 8;
|
||||
|
||||
uint8_t ExpectedLower = Expected;
|
||||
uint8_t ExpectedUpper = Expected >> 8;
|
||||
|
||||
uint8_t ActualUpper{};
|
||||
uint8_t ActualLower{};
|
||||
// Careful ordering here
|
||||
ActualUpper = LoadAcquire8(AddrUpper);
|
||||
ActualLower = LoadAcquire8(Addr);
|
||||
if (ActualUpper == ExpectedUpper &&
|
||||
ActualLower == ExpectedLower) {
|
||||
if (StoreCAS8(ExpectedUpper, DesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS8(ExpectedLower, DesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
}
|
||||
}
|
||||
|
||||
ActualLower = ExpectedLower;
|
||||
ActualUpper = ExpectedUpper;
|
||||
}
|
||||
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint16_t FailedResult = ActualUpper;
|
||||
FailedResult <<= 8;
|
||||
FailedResult |= ActualLower;
|
||||
mcontext->regs[ExpectedReg] = FailedResult;
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
AlignmentMask = 0b111;
|
||||
if ((Addr & AlignmentMask) == 7) {
|
||||
// Crosses 8byte boundary
|
||||
// Needs 128bit CAS
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t> *Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
|
||||
__uint128_t Mask = ~0U;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = mcontext->regs[DesiredReg] & 0xFFFF;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = mcontext->regs[ExpectedReg] & 0xFFFF;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg] = FailedResult;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
AlignmentMask = 0b11;
|
||||
if ((Addr & AlignmentMask) == 3) {
|
||||
// Crosses 4byte boundary
|
||||
// Needs 64bit CAS
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
|
||||
uint64_t Mask = 0xFFFF;
|
||||
Mask <<= Alignment * 8;
|
||||
|
||||
uint64_t NegMask = ~Mask;
|
||||
|
||||
uint64_t TmpExpected{};
|
||||
uint64_t TmpDesired{};
|
||||
|
||||
uint64_t Desired = mcontext->regs[DesiredReg] & 0xFFFF;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
uint64_t Expected = mcontext->regs[ExpectedReg] & 0xFFFF;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
std::atomic<uint64_t> *Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
while (1) {
|
||||
TmpExpected = Atomic->load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we can try again
|
||||
uint64_t FailedResultOurBits = TmpExpected & Mask;
|
||||
uint64_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
uint64_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
uint64_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg] = FailedResult;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Fits within 4byte boundary
|
||||
// Only needs 32bit CAS
|
||||
// Only alignment offset will be 1 here
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
|
||||
uint32_t Mask = 0xFFFF;
|
||||
Mask <<= Alignment * 8;
|
||||
|
||||
uint32_t NegMask = ~Mask;
|
||||
|
||||
uint32_t TmpExpected{};
|
||||
uint32_t TmpDesired{};
|
||||
|
||||
uint32_t Desired = mcontext->regs[DesiredReg] & 0xFFFF;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
uint32_t Expected = mcontext->regs[ExpectedReg] & 0xFFFF;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
std::atomic<uint32_t> *Atomic = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
while (1) {
|
||||
TmpExpected = Atomic->load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we can try again
|
||||
uint32_t FailedResultOurBits = TmpExpected & Mask;
|
||||
uint32_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
uint32_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
uint32_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg] = FailedResult;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (Size == 4) {
|
||||
// 32bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 12) {
|
||||
// Address crosses over 16byte threshold
|
||||
// Needs dual 4 byte CAS loop
|
||||
uint64_t Alignment = Addr & 0b11;
|
||||
Addr &= ~0b11;
|
||||
|
||||
uint64_t AddrUpper = Addr + 4;
|
||||
|
||||
uint64_t Desired = mcontext->regs[DesiredReg] & ~0U;
|
||||
uint64_t Expected = mcontext->regs[ExpectedReg] & ~0U;
|
||||
|
||||
Desired <<= Alignment * 8;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
uint64_t Mask = ~0U;
|
||||
Mask <<= Alignment * 8;
|
||||
uint64_t NegMask = ~Mask;
|
||||
|
||||
// Careful ordering here
|
||||
while (1) {
|
||||
uint64_t LoadOrderUpper = LoadAcquire32(AddrUpper);
|
||||
LoadOrderUpper <<= 32;
|
||||
uint64_t TmpActual = LoadOrderUpper | LoadAcquire32(Addr);
|
||||
|
||||
uint64_t TmpExpected = TmpActual;
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
uint64_t TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
if (TmpExpected == TmpActual) {
|
||||
uint32_t TmpExpectedLower = TmpExpected;
|
||||
uint32_t TmpExpectedUpper = TmpExpected >> 32;
|
||||
|
||||
uint32_t TmpDesiredLower = TmpDesired;
|
||||
uint32_t TmpDesiredUpper = TmpDesired >> 32;
|
||||
|
||||
if (StoreCAS32(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS32(TmpExpectedLower, TmpDesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
}
|
||||
}
|
||||
|
||||
TmpExpected = TmpExpectedUpper;
|
||||
TmpExpected <<= 32;
|
||||
TmpExpected |= TmpExpectedLower;
|
||||
}
|
||||
else {
|
||||
// Mismatch up front
|
||||
TmpExpected = TmpActual;
|
||||
}
|
||||
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
uint64_t FailedResultOurBits = TmpExpected & Mask;
|
||||
uint64_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
uint64_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
uint64_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg] = FailedResult;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
AlignmentMask = 0b111;
|
||||
if ((Addr & AlignmentMask) >= 5) {
|
||||
// Crosses 8byte boundary
|
||||
// Needs 128bit CAS
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t> *Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
|
||||
__uint128_t Mask = ~0U;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = mcontext->regs[DesiredReg] & ~0U;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = mcontext->regs[ExpectedReg] & ~0U;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg] = FailedResult;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Fits within 8byte boundary
|
||||
// Only needs 64bit CAS
|
||||
// Alignments can be [1,5)
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
|
||||
uint64_t Mask = ~0U;
|
||||
Mask <<= Alignment * 8;
|
||||
|
||||
uint64_t NegMask = ~Mask;
|
||||
|
||||
uint64_t TmpExpected{};
|
||||
uint64_t TmpDesired{};
|
||||
|
||||
uint64_t Desired = mcontext->regs[DesiredReg] & ~0U;
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
uint64_t Expected = mcontext->regs[ExpectedReg] & ~0U;
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
std::atomic<uint64_t> *Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
while (1) {
|
||||
TmpExpected = Atomic->load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we can try again
|
||||
uint64_t FailedResultOurBits = TmpExpected & Mask;
|
||||
uint64_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
uint64_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
uint64_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg] = FailedResult;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (Size == 8) {
|
||||
// 64bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
uint64_t Alignment = Addr & 0b111;
|
||||
Addr &= ~0b111ULL;
|
||||
uint64_t AddrUpper = Addr + 8;
|
||||
|
||||
// Crosses a 16byte boundary
|
||||
// Need to do 256bit atomic, but since that doesn't exist we need to do a dual CAS loop
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = mcontext->regs[DesiredReg];
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = mcontext->regs[ExpectedReg];
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
__uint128_t LoadOrderUpper = LoadAcquire64(AddrUpper);
|
||||
LoadOrderUpper <<= 64;
|
||||
__uint128_t TmpActual = LoadOrderUpper | LoadAcquire64(Addr);
|
||||
|
||||
// Set up expected
|
||||
TmpExpected = TmpActual;
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
uint64_t TmpExpectedLower = TmpExpected;
|
||||
uint64_t TmpExpectedUpper = TmpExpected >> 64;
|
||||
|
||||
uint64_t TmpDesiredLower = TmpDesired;
|
||||
uint64_t TmpDesiredUpper = TmpDesired >> 64;
|
||||
|
||||
if (TmpExpected == TmpActual) {
|
||||
if (StoreCAS64(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS64(TmpExpectedLower, TmpDesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
}
|
||||
}
|
||||
|
||||
TmpExpected = TmpExpectedUpper;
|
||||
TmpExpected <<= 64;
|
||||
TmpExpected |= TmpExpectedLower;
|
||||
}
|
||||
else {
|
||||
// Mismatch up front
|
||||
TmpExpected = TmpActual;
|
||||
}
|
||||
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg] = FailedResult;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
std::atomic<__uint128_t> *Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
__uint128_t NegMask = ~Mask;
|
||||
__uint128_t TmpExpected{};
|
||||
__uint128_t TmpDesired{};
|
||||
|
||||
__uint128_t Desired = mcontext->regs[DesiredReg];
|
||||
Desired <<= Alignment * 8;
|
||||
|
||||
__uint128_t Expected = mcontext->regs[ExpectedReg];
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
TmpExpected |= Expected;
|
||||
|
||||
// Set up desired
|
||||
TmpDesired = TmpExpected;
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg] = FailedResult;
|
||||
return true;
|
||||
}
|
||||
|
||||
// If we got here, that means the CAS failed
|
||||
// NotOurBits didn't change and bits we cared about didn't change
|
||||
ERROR_AND_DIE("Impossible");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
#pragma once
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t CASPAL_MASK = 0xBF'E0'FC'00;
|
||||
constexpr uint32_t CASPAL_INST = 0x08'60'FC'00;
|
||||
|
||||
constexpr uint32_t CASAL_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t CASAL_INST = 0x08'E0'FC'00;
|
||||
|
||||
bool HandleCASPAL(void *_mcontext, void *_info, uint32_t Instr);
|
||||
bool HandleCASAL(void *_mcontext, void *_info, uint32_t Instr);
|
||||
}
|
||||
+1
-2
@@ -327,8 +327,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h() {
|
||||
}
|
||||
|
||||
constexpr char ProcessorBrand[48] = {
|
||||
"FEX-"
|
||||
GIT_SHORT_HASH
|
||||
GIT_DESCRIBE_STRING
|
||||
"\0"
|
||||
};
|
||||
|
||||
|
||||
+19
-37
@@ -1,5 +1,5 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/BlockCache.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
@@ -33,7 +33,7 @@ namespace FEXCore {
|
||||
WorkerThread.join();
|
||||
}
|
||||
|
||||
void CompileService::ClearCache(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
void CompileService::ClearCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// On cache clear we need to spin down the execution thread to ensure it isn't trying to give us more work items
|
||||
if (CompileMutex.try_lock()) {
|
||||
// We can only clear these things if we pulled the compile mutex
|
||||
@@ -57,30 +57,19 @@ namespace FEXCore {
|
||||
}
|
||||
}
|
||||
|
||||
if (GuestRIP == 0) {
|
||||
CompileThreadData->IRLists.clear();
|
||||
}
|
||||
else {
|
||||
auto IR = CompileThreadData->IRLists.find(GuestRIP)->second.release();
|
||||
CompileThreadData->IRLists.clear();
|
||||
CompileThreadData->IRLists.try_emplace(GuestRIP, IR);
|
||||
}
|
||||
LogMan::Throw::A(CompileThreadData->IRLists.size() == 0, "Compile service must never have IRLists");
|
||||
LogMan::Throw::A(CompileThreadData->RALists.size() == 0, "Compile service must never have RALists");
|
||||
LogMan::Throw::A(CompileThreadData->DebugData.size() == 0, "Compile service must never have DebugData");
|
||||
|
||||
CompileMutex.unlock();
|
||||
}
|
||||
|
||||
// Clear the inverse cache of what is calling us from the Context ClearCache routine
|
||||
auto SelectedThread = Thread->IsCompileService ? ParentThread : Thread;
|
||||
SelectedThread->BlockCache->ClearCache();
|
||||
SelectedThread->LookupCache->ClearCache();
|
||||
SelectedThread->CPUBackend->ClearCache();
|
||||
SelectedThread->IntBackend->ClearCache();
|
||||
}
|
||||
|
||||
void CompileService::RemoveCodeEntry(uint64_t GuestRIP) {
|
||||
CompileThreadData->IRLists.erase(GuestRIP);
|
||||
CompileThreadData->DebugData.erase(GuestRIP);
|
||||
CompileThreadData->BlockCache->Erase(GuestRIP);
|
||||
}
|
||||
|
||||
CompileService::WorkItem *CompileService::CompileCode(uint64_t RIP) {
|
||||
// Tell the worker thread to compile code for us
|
||||
@@ -132,33 +121,26 @@ namespace FEXCore {
|
||||
|
||||
// If we had a work item then work on it
|
||||
if (Item) {
|
||||
// Does the block cache already contain this RIP?
|
||||
void *CompiledCode = reinterpret_cast<void*>(CompileThreadData->BlockCache->FindBlock(Item->RIP));
|
||||
FEXCore::Core::DebugData *DebugData = nullptr;
|
||||
// Make sure it's not in lookup cache by accident
|
||||
LogMan::Throw::A(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
|
||||
|
||||
// Code isn't in cache, compile now
|
||||
// Set our thread state's RIP
|
||||
CompileThreadData->State.State.rip = Item->RIP;
|
||||
|
||||
if (!CompiledCode) {
|
||||
// Code isn't in cache, compile now
|
||||
// Set our thread state's RIP
|
||||
CompileThreadData->State.State.rip = Item->RIP;
|
||||
auto [Code, Data] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
|
||||
CompiledCode = Code;
|
||||
DebugData = Data;
|
||||
}
|
||||
auto [CodePtr, IRList, DebugData, RAData, Generated] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
|
||||
|
||||
if (!CompiledCode) {
|
||||
LogMan::Throw::A(Generated == true, "Compile Service doesn't have IR Cache");
|
||||
|
||||
if (!CodePtr) {
|
||||
// XXX: We currently have the expectation that compile service code will be significantly smaller than regular thread's code
|
||||
ERROR_AND_DIE("Couldn't compile code for thread at RIP: 0x%lx", Item->RIP);
|
||||
}
|
||||
|
||||
auto BlockMapPtr = CompileThreadData->BlockCache->AddBlockMapping(Item->RIP, CompiledCode);
|
||||
if (BlockMapPtr == 0) {
|
||||
// XXX: We currently have the expectation that compiler service block cache will be significantly underutilized compared to regular thread
|
||||
ERROR_AND_DIE("Couldn't add code to block cache for thread at RIP: 0x%lx", Item->RIP);
|
||||
}
|
||||
|
||||
Item->CodePtr = CompiledCode;
|
||||
Item->IRList = CompileThreadData->IRLists.find(Item->RIP)->second.get();
|
||||
Item->CodePtr = CodePtr;
|
||||
Item->IRList = IRList;
|
||||
Item->DebugData = DebugData;
|
||||
Item->RAData = RAData;
|
||||
|
||||
GCArray.emplace_back(Item);
|
||||
Item->ServiceWorkDone.NotifyAll();
|
||||
|
||||
+5
-2
@@ -16,6 +16,9 @@ namespace Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace IR {
|
||||
class RegisterAllocationData;
|
||||
};
|
||||
class CompileService final {
|
||||
public:
|
||||
CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
@@ -29,6 +32,7 @@ class CompileService final {
|
||||
// Outgoing
|
||||
void *CodePtr{};
|
||||
FEXCore::IR::IRListView<true> *IRList{};
|
||||
FEXCore::IR::RegisterAllocationData *RAData{};
|
||||
FEXCore::Core::DebugData *DebugData{};
|
||||
|
||||
// Communication
|
||||
@@ -37,8 +41,7 @@ class CompileService final {
|
||||
};
|
||||
|
||||
WorkItem *CompileCode(uint64_t RIP);
|
||||
void ClearCache(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
void RemoveCodeEntry(uint64_t GuestRIP);
|
||||
void ClearCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
|
||||
+262
-234
@@ -2,7 +2,7 @@
|
||||
#include "Common/Paths.h"
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/BlockCache.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
@@ -111,7 +111,7 @@ namespace DefaultFallbackCore {
|
||||
void Initialize() override {}
|
||||
bool NeedsOpDispatch() override { return false; }
|
||||
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData) override {
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override {
|
||||
LogMan::Msg::E("Fell back to default code handler at RIP: 0x%lx", ThreadState->State.State.rip);
|
||||
return nullptr;
|
||||
}
|
||||
@@ -378,7 +378,7 @@ namespace FEXCore::Context {
|
||||
// Walk the threads and tell them to clear their caches
|
||||
// Useful when our block size is set to a large number and we need to step a single instruction
|
||||
for (auto &Thread : Threads) {
|
||||
ClearCodeCache(Thread, 0);
|
||||
ClearCodeCache(Thread, true);
|
||||
}
|
||||
}
|
||||
CoreRunningMode PreviousRunningMode = this->Config.RunningMode;
|
||||
@@ -457,13 +457,13 @@ namespace FEXCore::Context {
|
||||
|
||||
void Context::InitializeThreadData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CPUBackend->Initialize();
|
||||
Thread->IntBackend->Initialize();
|
||||
Thread->FallbackBackend->Initialize();
|
||||
|
||||
auto IRHandler = [Thread](uint64_t Addr, IR::IREmitter *IR) -> void {
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(IR);
|
||||
Thread->IRLists.try_emplace(Addr, IR->CreateIRCopy());
|
||||
Thread->DebugData.try_emplace(Addr, new Core::DebugData());
|
||||
Thread->RALists.try_emplace(Addr, Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->PullAllocationData() : nullptr);
|
||||
};
|
||||
|
||||
LocalLoader->AddIR(IRHandler);
|
||||
@@ -494,7 +494,7 @@ namespace FEXCore::Context {
|
||||
void Context::InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread) {
|
||||
State->OpDispatcher = std::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
State->OpDispatcher->SetMultiblock(Config.Multiblock);
|
||||
State->BlockCache = std::make_unique<FEXCore::BlockCache>(this);
|
||||
State->LookupCache = std::make_unique<FEXCore::LookupCache>(this);
|
||||
State->FrontendDecoder = std::make_unique<FEXCore::Frontend::Decoder>(this);
|
||||
State->PassManager = std::make_unique<FEXCore::IR::PassManager>();
|
||||
State->PassManager->RegisterExitHandler([this]() {
|
||||
@@ -518,23 +518,14 @@ namespace FEXCore::Context {
|
||||
switch (Config.Core) {
|
||||
case FEXCore::Config::CONFIG_INTERPRETER:
|
||||
State->CPUBackend.reset(FEXCore::CPU::CreateInterpreterCore(this, State, CompileThread));
|
||||
State->IntBackend = State->CPUBackend;
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
State->PassManager->InsertRegisterAllocationPass(DoSRA);
|
||||
// Initialization order matters here, the IR JIT may want to have the interpreter created first to get a pointer to its execution function
|
||||
// This is useful for JIT to interpreter fallback support
|
||||
State->IntBackend.reset(FEXCore::CPU::CreateInterpreterCore(this, State, CompileThread));
|
||||
State->CPUBackend.reset(FEXCore::CPU::CreateJITCore(this, State, CompileThread));
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM: State->CPUBackend.reset(CustomCPUFactory(this, &State->State)); break;
|
||||
default: LogMan::Msg::A("Unknown core configuration");
|
||||
}
|
||||
|
||||
if (!State->IntBackend) {
|
||||
State->IntBackend.reset(FEXCore::CPU::CreateInterpreterCore(this, State, CompileThread));
|
||||
}
|
||||
State->FallbackBackend.reset(FallbackCPUFactory(this, &State->State));
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* Context::CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
@@ -555,299 +546,339 @@ namespace FEXCore::Context {
|
||||
|
||||
InitializeCompiler(Thread, false);
|
||||
|
||||
LogMan::Throw::A(!Thread->FallbackBackend->NeedsOpDispatch(), "Fallback CPU backend must not require OpDispatch");
|
||||
return Thread;
|
||||
}
|
||||
|
||||
uintptr_t Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
auto BlockMapPtr = Thread->BlockCache->AddBlockMapping(Address, Ptr);
|
||||
if (BlockMapPtr == 0) {
|
||||
Thread->BlockCache->ClearCache();
|
||||
|
||||
// Pull out the current IR we added and store it back after we cleared the rest of the list
|
||||
// Needed in the case the the block mapping has aliased
|
||||
auto iter = Thread->IRLists.find(Address);
|
||||
if (iter != Thread->IRLists.end()) {
|
||||
auto IR = iter->second.release();
|
||||
Thread->IRLists.clear();
|
||||
Thread->IRLists.try_emplace(Address, IR);
|
||||
}
|
||||
BlockMapPtr = Thread->BlockCache->AddBlockMapping(Address, Ptr);
|
||||
LogMan::Throw::A(BlockMapPtr, "Couldn't add mapping after clearing mapping cache");
|
||||
}
|
||||
|
||||
return BlockMapPtr;
|
||||
void Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
Thread->BlockCache->ClearCache();
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache) {
|
||||
Thread->LookupCache->ClearCache();
|
||||
Thread->CPUBackend->ClearCache();
|
||||
Thread->IntBackend->ClearCache();
|
||||
|
||||
if (Thread->CompileService) {
|
||||
Thread->CompileService->ClearCache(Thread, GuestRIP);
|
||||
Thread->CompileService->ClearCache(Thread);
|
||||
}
|
||||
|
||||
if (GuestRIP == 0) {
|
||||
if (AlsoClearIRCache) {
|
||||
Thread->IRLists.clear();
|
||||
}
|
||||
else {
|
||||
auto IR = Thread->IRLists.find(GuestRIP)->second.release();
|
||||
Thread->IRLists.clear();
|
||||
Thread->IRLists.try_emplace(GuestRIP, IR);
|
||||
Thread->RALists.clear();
|
||||
Thread->DebugData.clear();
|
||||
}
|
||||
}
|
||||
|
||||
std::tuple<void *, FEXCore::Core::DebugData *> Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
std::tuple<FEXCore::IR::IRListView<true> *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t> Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
uint8_t const *GuestCode{};
|
||||
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
|
||||
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
|
||||
|
||||
// Do we already have this in the IR cache?
|
||||
auto IR = Thread->IRLists.find(GuestRIP);
|
||||
FEXCore::IR::IRListView<true> *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
bool HadDispatchError {false};
|
||||
|
||||
if (IR == Thread->IRLists.end()) {
|
||||
bool HadDispatchError {false};
|
||||
uint64_t TotalInstructions {0};
|
||||
uint64_t TotalInstructionsLength {0};
|
||||
|
||||
uint64_t TotalInstructions {0};
|
||||
uint64_t TotalInstructionsLength {0};
|
||||
if (!Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP)) {
|
||||
if (Config.BreakOnFrontendFailure) {
|
||||
LogMan::Msg::E("Had Frontend decoder error");
|
||||
Stop(false /* Ignore Current Thread */);
|
||||
}
|
||||
return { nullptr, nullptr, 0, 0 };
|
||||
}
|
||||
|
||||
if (!Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP)) {
|
||||
if (Config.BreakOnFrontendFailure) {
|
||||
LogMan::Msg::E("Had Frontend decoder error");
|
||||
Stop(false /* Ignore Current Thread */);
|
||||
}
|
||||
return { nullptr, nullptr };
|
||||
auto CodeBlocks = Thread->FrontendDecoder->GetDecodedBlocks();
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks);
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
FEXCore::Frontend::Decoder::DecodedBlocks const &Block = CodeBlocks->at(j);
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
|
||||
|
||||
uint64_t BlockInstructionsLength {};
|
||||
|
||||
// Reset any block-specific state
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
|
||||
if (Block.HasInvalidInstruction) {
|
||||
uint8_t GPRSize = Config.Is64BitMode ? 8 : 4;
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_Constant(GPRSize * 8, Block.Entry));
|
||||
break;
|
||||
}
|
||||
|
||||
auto CodeBlocks = Thread->FrontendDecoder->GetDecodedBlocks();
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
FEXCore::X86Tables::X86InstInfo const* TableInfo {nullptr};
|
||||
FEXCore::X86Tables::DecodedInst const* DecodedInfo {nullptr};
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks);
|
||||
TableInfo = Block.DecodedInstructions[i].TableInfo;
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
FEXCore::Frontend::Decoder::DecodedBlocks const &Block = CodeBlocks->at(j);
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
if (Config.SMCChecks) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint64_t*>(Block.Entry + BlockInstructionsLength);
|
||||
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(ExistingCodePtr[0], ExistingCodePtr[1], (uintptr_t)ExistingCodePtr, DecodedInfo->InstSize);
|
||||
|
||||
uint64_t BlockInstructionsLength {};
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->_CondJump(CodeChanged);
|
||||
|
||||
// Reset any block-specific state
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
auto CurrentBlock = Thread->OpDispatcher->GetCurrentBlock();
|
||||
auto CodeWasChangedBlock = Thread->OpDispatcher->CreateNewCodeBlockAtEnd();
|
||||
Thread->OpDispatcher->SetTrueJumpTarget(InvalidateCodeCond, CodeWasChangedBlock);
|
||||
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
FEXCore::X86Tables::X86InstInfo const* TableInfo {nullptr};
|
||||
FEXCore::X86Tables::DecodedInst const* DecodedInfo {nullptr};
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_RemoveCodeEntry(GuestRIP);
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_Constant(Block.Entry + BlockInstructionsLength));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
TableInfo = Block.DecodedInstructions[i].TableInfo;
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
Thread->OpDispatcher->SetFalseJumpTarget(InvalidateCodeCond, NextOpBlock);
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
|
||||
}
|
||||
|
||||
if (Config.SMCChecks) {
|
||||
__uint128_t existing;
|
||||
|
||||
uintptr_t ExistingCodePtr{};
|
||||
|
||||
ExistingCodePtr = reinterpret_cast<uintptr_t>(Block.Entry + BlockInstructionsLength);
|
||||
|
||||
memcpy(&existing, (void*)(ExistingCodePtr), DecodedInfo->InstSize);
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(existing, ExistingCodePtr, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->_CondJump(CodeChanged);
|
||||
|
||||
auto CurrentBlock = Thread->OpDispatcher->GetCurrentBlock();
|
||||
auto CodeWasChangedBlock = Thread->OpDispatcher->CreateNewCodeBlockAtEnd();
|
||||
Thread->OpDispatcher->SetTrueJumpTarget(InvalidateCodeCond, CodeWasChangedBlock);
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_RemoveCodeEntry(GuestRIP);
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_Constant(Block.Entry + BlockInstructionsLength));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
Thread->OpDispatcher->SetFalseJumpTarget(InvalidateCodeCond, NextOpBlock);
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
|
||||
}
|
||||
|
||||
if (TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
if (Config.BreakOnFrontendFailure) {
|
||||
LogMan::Msg::E("Had OpDispatcher error at 0x%lx", GuestRIP);
|
||||
Stop(false /* Ignore Current Thread */);
|
||||
}
|
||||
HadDispatchError = true;
|
||||
if (TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
if (Config.BreakOnFrontendFailure) {
|
||||
LogMan::Msg::E("Had OpDispatcher error at 0x%lx", GuestRIP);
|
||||
Stop(false /* Ignore Current Thread */);
|
||||
}
|
||||
else {
|
||||
BlockInstructionsLength += DecodedInfo->InstSize;
|
||||
TotalInstructionsLength += DecodedInfo->InstSize;
|
||||
++TotalInstructions;
|
||||
}
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Missing OpDispatcher at 0x%lx{'%s'}", Block.Entry + BlockInstructionsLength, TableInfo->Name);
|
||||
HadDispatchError = true;
|
||||
}
|
||||
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError) {
|
||||
if (TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return { nullptr, nullptr };
|
||||
}
|
||||
else {
|
||||
uint8_t GPRSize = Config.Is64BitMode ? 8 : 4;
|
||||
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_Constant(GPRSize * 8, Block.Entry + BlockInstructionsLength));
|
||||
break;
|
||||
}
|
||||
else {
|
||||
BlockInstructionsLength += DecodedInfo->InstSize;
|
||||
TotalInstructionsLength += DecodedInfo->InstSize;
|
||||
++TotalInstructions;
|
||||
}
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Missing OpDispatcher at 0x%lx{'%s'}", Block.Entry + BlockInstructionsLength, TableInfo->Name);
|
||||
HadDispatchError = true;
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->FinishOp(DecodedInfo->PC + DecodedInfo->InstSize, i + 1 == InstsInBlock)) {
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError) {
|
||||
if (TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return { nullptr, nullptr, 0, 0 };
|
||||
}
|
||||
else {
|
||||
uint8_t GPRSize = Config.Is64BitMode ? 8 : 4;
|
||||
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_Constant(GPRSize * 8, Block.Entry + BlockInstructionsLength));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->FinishOp(DecodedInfo->PC + DecodedInfo->InstSize, i + 1 == InstsInBlock)) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
auto IRDumper = [Thread, GuestRIP](IR::RegisterAllocationData* RA) {
|
||||
FILE* f = nullptr;
|
||||
bool CloseAfter = false;
|
||||
|
||||
if (Thread->CTX->Config.DumpIR=="stderr") {
|
||||
f = stderr;
|
||||
}
|
||||
else if (Thread->CTX->Config.DumpIR=="stdout") {
|
||||
f = stdout;
|
||||
}
|
||||
else {
|
||||
std::stringstream fileName;
|
||||
fileName << Thread->CTX->Config.DumpIR << "/" << std::hex << GuestRIP << (RA ? "-post.ir" : "-pre.ir");
|
||||
|
||||
f = fopen(fileName.str().c_str(), "w");
|
||||
CloseAfter = true;
|
||||
}
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
|
||||
|
||||
auto IRDumper = [Thread, GuestRIP](IR::RegisterAllocationPass* RA) {
|
||||
FILE* f = nullptr;
|
||||
bool CloseAfter = false;
|
||||
|
||||
if (Thread->CTX->Config.DumpIR=="stderr") {
|
||||
f = stderr;
|
||||
}
|
||||
else if (Thread->CTX->Config.DumpIR=="stdout") {
|
||||
f = stdout;
|
||||
}
|
||||
else {
|
||||
std::stringstream fileName;
|
||||
fileName << Thread->CTX->Config.DumpIR << "/" << std::hex << GuestRIP << (RA ? "-post.ir" : "-pre.ir");
|
||||
|
||||
f = fopen(fileName.str().c_str(), "w");
|
||||
CloseAfter = true;
|
||||
}
|
||||
|
||||
if (f) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fprintf(f,"IR-%s 0x%lx:\n%s\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str().c_str());
|
||||
|
||||
if (CloseAfter) {
|
||||
fclose(f);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (Thread->CTX->Config.DumpIR != "no") {
|
||||
IRDumper(nullptr);
|
||||
}
|
||||
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(Thread->OpDispatcher.get());
|
||||
|
||||
if (Thread->CTX->Config.DumpIR != "no") {
|
||||
IRDumper(Thread->PassManager->GetRAPass());
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->ShouldDump) {
|
||||
if (f) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, Thread->PassManager->GetRAPass());
|
||||
printf("IR 0x%lx:\n%s\n@@@@@\n", GuestRIP, out.str().c_str());
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fprintf(f,"IR-%s 0x%lx:\n%s\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str().c_str());
|
||||
|
||||
if (CloseAfter) {
|
||||
fclose(f);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Create a copy of the IR and place it in this thread's IR cache
|
||||
auto AddedIR = Thread->IRLists.try_emplace(GuestRIP, Thread->OpDispatcher->CreateIRCopy());
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
auto Debugit = &Thread->DebugData.try_emplace(GuestRIP).first->second;
|
||||
Debugit->GuestCodeSize = TotalInstructionsLength;
|
||||
Debugit->GuestInstructionCount = TotalInstructions;
|
||||
|
||||
IRList = AddedIR.first->second.get();
|
||||
DebugData = Debugit;
|
||||
Thread->Stats.BlocksCompiled.fetch_add(1);
|
||||
if (Thread->CTX->Config.DumpIR != "no") {
|
||||
IRDumper(nullptr);
|
||||
}
|
||||
else {
|
||||
IRList = IR->second.get();
|
||||
auto Debugit = Thread->DebugData.find(GuestRIP);
|
||||
if (Debugit != Thread->DebugData.end()) {
|
||||
DebugData = &Debugit->second;
|
||||
|
||||
if (Thread->CTX->Config.ValidateIRarser) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
std::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction();
|
||||
compaction->Run(Thread->OpDispatcher.get());
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
auto reparsed = IR::Parse(&out);
|
||||
if (reparsed == nullptr) {
|
||||
LogMan::Msg::A("Failed to parse ir\n");
|
||||
} else {
|
||||
std::stringstream out2;
|
||||
auto NewIR2 = reparsed->ViewIR();
|
||||
Dump(&out2, &NewIR2, nullptr);
|
||||
if (out.str() != out2.str()) {
|
||||
printf("one:\n %s\n", out.str().c_str());
|
||||
printf("two:\n %s\n", out2.str().c_str());
|
||||
LogMan::Msg::A("Parsed ir doesn't match\n");
|
||||
}
|
||||
delete reparsed;
|
||||
}
|
||||
}
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(Thread->OpDispatcher.get());
|
||||
|
||||
if (Thread->CTX->Config.DumpIR != "no") {
|
||||
IRDumper(Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->GetAllocationData() : nullptr);
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->ShouldDump) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->GetAllocationData() : nullptr);
|
||||
printf("IR 0x%lx:\n%s\n@@@@@\n", GuestRIP, out.str().c_str());
|
||||
}
|
||||
|
||||
auto RAData = Thread->PassManager->GetRAPass() ? Thread->PassManager->GetRAPass()->PullAllocationData() : nullptr;
|
||||
auto IRList = Thread->OpDispatcher->CreateIRCopy();
|
||||
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
return {IRList, RAData.release(), TotalInstructions, TotalInstructionsLength};
|
||||
}
|
||||
|
||||
std::tuple<void *, FEXCore::IR::IRListView<true> *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool> Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
FEXCore::IR::IRListView<true> *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
bool GeneratedIR {};
|
||||
|
||||
// Do we already have this in the IR cache?
|
||||
auto IR = Thread->IRLists.find(GuestRIP);
|
||||
|
||||
if (IR != Thread->IRLists.end()) {
|
||||
// Entry already exists
|
||||
// pull in the data
|
||||
IRList = IR->second.get();
|
||||
DebugData = Thread->DebugData.find(GuestRIP)->second.get();
|
||||
RAData = Thread->RALists.find(GuestRIP)->second.get();
|
||||
|
||||
GeneratedIR = false;
|
||||
} else {
|
||||
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength] = GenerateIR(Thread, GuestRIP);
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
RAData = RACopy;
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
|
||||
// Initialize metadata
|
||||
DebugData->GuestCodeSize = TotalInstructionsLength;
|
||||
DebugData->GuestInstructionCount = TotalInstructions;
|
||||
|
||||
// Increment stats
|
||||
Thread->Stats.BlocksCompiled.fetch_add(1);
|
||||
|
||||
// These blocks aren't already in the cache
|
||||
GeneratedIR = true;
|
||||
}
|
||||
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
return { Thread->CPUBackend->CompileCode(IRList, DebugData), DebugData };
|
||||
return { Thread->CPUBackend->CompileCode(IRList, DebugData, RAData), IRList, DebugData, RAData, GeneratedIR};
|
||||
}
|
||||
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
void *CodePtr;
|
||||
FEXCore::Core::DebugData *DebugData;
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void *CodePtr {};
|
||||
FEXCore::IR::IRListView<true> *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
|
||||
bool DecrementRefCount = false;
|
||||
bool GeneratedIR {};
|
||||
|
||||
if (Thread->CompileBlockReentrantRefCount != 0) {
|
||||
if (!Thread->CompileService) {
|
||||
Thread->CompileService = std::make_shared<FEXCore::CompileService>(this, Thread);
|
||||
Thread->CompileService->Initialize();
|
||||
}
|
||||
|
||||
auto WorkItem = Thread->CompileService->CompileCode(GuestRIP);
|
||||
WorkItem->ServiceWorkDone.Wait();
|
||||
// Return here with the data in place
|
||||
Thread->IRLists.try_emplace(GuestRIP, WorkItem->IRList->CreateCopy());
|
||||
CodePtr = WorkItem->CodePtr;
|
||||
IRList = WorkItem->IRList;
|
||||
DebugData = WorkItem->DebugData;
|
||||
RAData = WorkItem->RAData;
|
||||
WorkItem->SafeToClear = true;
|
||||
|
||||
// The compile service will always generate IR + DebugData + RAData
|
||||
// Remove the entries here to make sure we don't fail to insert later on
|
||||
RemoveCodeEntry(Thread, GuestRIP);
|
||||
GeneratedIR = true;
|
||||
} else {
|
||||
++Thread->CompileBlockReentrantRefCount;
|
||||
DecrementRefCount = true;
|
||||
auto [Code, Data] = CompileCode(Thread, GuestRIP);
|
||||
auto [Code, IR, Data, RA, Generated] = CompileCode(Thread, GuestRIP);
|
||||
CodePtr = Code;
|
||||
IRList = IR;
|
||||
DebugData = Data;
|
||||
RAData = RA;
|
||||
GeneratedIR = Generated;
|
||||
}
|
||||
|
||||
if (CodePtr != nullptr) {
|
||||
// The core managed to compile the code.
|
||||
LogMan::Throw::A(CodePtr != nullptr, "Failed to compile code %lX", GuestRIP);
|
||||
|
||||
// The core managed to compile the code.
|
||||
#if ENABLE_JITSYMBOLS
|
||||
if (DebugData) {
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
Symbols.Register((void*)Subblock.HostCodeStart, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
} else {
|
||||
Symbols.Register(CodePtr, GuestRIP, DebugData->HostCodeSize);
|
||||
if (DebugData) {
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
Symbols.Register((void*)Subblock.HostCodeStart, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
} else {
|
||||
Symbols.Register(CodePtr, GuestRIP, DebugData->HostCodeSize);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
if (DecrementRefCount)
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
return AddBlockMapping(Thread, GuestRIP, CodePtr);
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
Thread->IRLists.emplace(GuestRIP, IRList);
|
||||
Thread->DebugData.emplace(GuestRIP, DebugData);
|
||||
Thread->RALists.emplace(GuestRIP, RAData);
|
||||
}
|
||||
|
||||
if (DecrementRefCount)
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
|
||||
return 0;
|
||||
}
|
||||
// Insert to lookup cache
|
||||
AddBlockMapping(Thread, GuestRIP, CodePtr);
|
||||
|
||||
uintptr_t Context::CompileFallbackBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
// We have ONE more chance to try and fallback to the fallback CPU backend
|
||||
// This will most likely fail since regular code use won't be using a fallback core.
|
||||
// It's mainly for testing new instruction encodings
|
||||
void *CodePtr = Thread->FallbackBackend->CompileCode(nullptr, nullptr);
|
||||
if (CodePtr) {
|
||||
uintptr_t Ptr = reinterpret_cast<uintptr_t >(AddBlockMapping(Thread, GuestRIP, CodePtr));
|
||||
return Ptr;
|
||||
}
|
||||
return (uintptr_t)CodePtr;
|
||||
|
||||
if (DecrementRefCount)
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -904,12 +935,9 @@ namespace FEXCore::Context {
|
||||
|
||||
void Context::RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
Thread->IRLists.erase(GuestRIP);
|
||||
Thread->RALists.erase(GuestRIP);
|
||||
Thread->DebugData.erase(GuestRIP);
|
||||
Thread->BlockCache->Erase(GuestRIP);
|
||||
|
||||
if (Thread->CompileService) {
|
||||
Thread->CompileService->RemoveCodeEntry(GuestRIP);
|
||||
}
|
||||
Thread->LookupCache->Erase(GuestRIP);
|
||||
}
|
||||
|
||||
// Debug interface
|
||||
@@ -920,7 +948,7 @@ namespace FEXCore::Context {
|
||||
// Erase the RIP from all the storage backings if it exists
|
||||
Thread->IRLists.erase(RIP);
|
||||
Thread->DebugData.erase(RIP);
|
||||
Thread->BlockCache->Erase(RIP);
|
||||
Thread->LookupCache->Erase(RIP);
|
||||
|
||||
// We don't care if compilation passes or not
|
||||
CompileBlock(Thread, RIP);
|
||||
@@ -951,7 +979,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
bool Context::FindHostCodeForRIP(uint64_t RIP, uint8_t **Code) {
|
||||
uintptr_t HostCode = ParentThread->BlockCache->FindBlock(RIP);
|
||||
uintptr_t HostCode = ParentThread->LookupCache->FindBlock(RIP);
|
||||
if (!HostCode) {
|
||||
return false;
|
||||
}
|
||||
|
||||
+73
-60
@@ -329,9 +329,21 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
// XXX: Once we support 32bit x86 then this will be necessary to support
|
||||
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_LEGACY_PREFIX, "Legacy Prefix");
|
||||
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_UNKNOWN, "Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_INVALID, "Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
|
||||
LogMan::Msg::D("Legacy Prefix");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
LogMan::Throw::A(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
"Group Ops should have been decoded before this!");
|
||||
|
||||
@@ -565,9 +577,22 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
// XXX: Once we support 32bit x86 then this will be necessary to support
|
||||
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_LEGACY_PREFIX, "Legacy Prefix");
|
||||
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_UNKNOWN, "Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_INVALID, "Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
|
||||
LogMan::Msg::D("Legacy Prefix");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
|
||||
|
||||
if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
|
||||
Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
|
||||
@@ -686,36 +711,12 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
return NormalOp(LocalInfo, Op);
|
||||
}
|
||||
else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
uint8_t P1 = ReadByte();
|
||||
uint8_t P2 = ReadByte();
|
||||
uint8_t P3 = ReadByte();
|
||||
/* uint8_t P1 = */ ReadByte();
|
||||
/* uint8_t P2 = */ ReadByte();
|
||||
/* uint8_t P3 = */ ReadByte();
|
||||
uint8_t EVEXOp = ReadByte();
|
||||
return NormalOp(&EVEXTableOps[EVEXOp], EVEXOp);
|
||||
}
|
||||
else if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
|
||||
LogMan::Throw::A(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
|
||||
|
||||
// Widening displacement
|
||||
if (Op & 0b1000) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_WIDENING;
|
||||
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_WIDENING_SIZE_LAST);
|
||||
}
|
||||
|
||||
// XGPR_B bit set
|
||||
if (Op & 0b0001)
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
|
||||
|
||||
// XGPR_X bit set
|
||||
if (Op & 0b0010)
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
|
||||
// XGPR_R bit set
|
||||
if (Op & 0b0100)
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
return NormalOp(Info, Op);
|
||||
}
|
||||
@@ -723,13 +724,12 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
InstructionSize = 0;
|
||||
Instruction.fill(0);
|
||||
bool InstructionDecoded = false;
|
||||
|
||||
DecodeInst = &DecodedBuffer[DecodedSize];
|
||||
memset(DecodeInst, 0, sizeof(DecodedInst));
|
||||
DecodeInst->PC = PC;
|
||||
|
||||
while (!InstructionDecoded) {
|
||||
for(;;) {
|
||||
uint8_t Op = ReadByte();
|
||||
switch (Op) {
|
||||
case 0x0F: {// Escape Op
|
||||
@@ -753,9 +753,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
|
||||
// Take a peek at the op just past the displacement
|
||||
uint8_t LocalOp = ReadByte();
|
||||
if (NormalOpHeader(&FEXCore::X86Tables::DDDNowOps[LocalOp], LocalOp)) {
|
||||
InstructionDecoded = true;
|
||||
}
|
||||
return NormalOpHeader(&FEXCore::X86Tables::DDDNowOps[LocalOp], LocalOp);
|
||||
break;
|
||||
}
|
||||
case 0x38: { // F38 Table!
|
||||
@@ -770,9 +768,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
Prefix = PF_38_66;
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
if (NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp)) {
|
||||
InstructionDecoded = true;
|
||||
}
|
||||
return NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp);
|
||||
break;
|
||||
}
|
||||
case 0x3A: { // F3A Table!
|
||||
@@ -788,9 +784,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
Prefix |= PF_3A_REX;
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
if (NormalOpHeader(&FEXCore::X86Tables::H0F3ATableOps[LocalOp], LocalOp)) {
|
||||
InstructionDecoded = true;
|
||||
}
|
||||
return NormalOpHeader(&FEXCore::X86Tables::H0F3ATableOps[LocalOp], LocalOp);
|
||||
break;
|
||||
}
|
||||
default: // Two byte table!
|
||||
@@ -805,37 +799,29 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
bool NoOverlay66 = (FEXCore::X86Tables::SecondBaseOps[EscapeOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
|
||||
|
||||
if (NoOverlay) { // This section of the table ignores prefix extention
|
||||
if (NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp)) {
|
||||
InstructionDecoded = true;
|
||||
}
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
|
||||
if (NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp)) {
|
||||
InstructionDecoded = true;
|
||||
}
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
|
||||
if (NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp)) {
|
||||
InstructionDecoded = true;
|
||||
}
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
DecodeFlags::PopOpAddrIf(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
|
||||
if (NormalOpHeader(&FEXCore::X86Tables::OpSizeModOps[EscapeOp], EscapeOp)) {
|
||||
InstructionDecoded = true;
|
||||
}
|
||||
return NormalOpHeader(&FEXCore::X86Tables::OpSizeModOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp)) {
|
||||
InstructionDecoded = true;
|
||||
else {
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -891,9 +877,33 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_GS_PREFIX;
|
||||
break;
|
||||
default: { // Default base table
|
||||
if (NormalOpHeader(&FEXCore::X86Tables::BaseOps[Op], Op)) {
|
||||
InstructionDecoded = true;
|
||||
auto Info = &FEXCore::X86Tables::BaseOps[Op];
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
|
||||
LogMan::Throw::A(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
|
||||
|
||||
// Widening displacement
|
||||
if (Op & 0b1000) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_WIDENING;
|
||||
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_WIDENING_SIZE_LAST);
|
||||
}
|
||||
|
||||
// XGPR_B bit set
|
||||
if (Op & 0b0001)
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
|
||||
|
||||
// XGPR_X bit set
|
||||
if (Op & 0b0010)
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
|
||||
// XGPR_R bit set
|
||||
if (Op & 0b0100)
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
|
||||
} else {
|
||||
return NormalOpHeader(Info, Op);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -1002,8 +1012,11 @@ bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
|
||||
while (1) {
|
||||
ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) {
|
||||
LogMan::Msg::D("Couldn't Decode something at 0x%lx, Started at 0x%lx", PC + PCOffset, PC);
|
||||
LogMan::Throw::A(EntryPoint != (RIPToDecode + PCOffset), "Trying to execute invalid code");
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
@@ -20,6 +20,7 @@ public:
|
||||
uint64_t Entry{};
|
||||
uint64_t NumInstructions{};
|
||||
FEXCore::X86Tables::DecodedInst *DecodedInstructions;
|
||||
bool HasInvalidInstruction{};
|
||||
};
|
||||
|
||||
Decoder(FEXCore::Context::Context *ctx);
|
||||
|
||||
@@ -199,13 +199,13 @@ DispatchGenerator::DispatchGenerator(FEXCore::Context::Context *ctx, FEXCore::Co
|
||||
auto RipReg = x2;
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
LoadConstant(x3, Thread->BlockCache->GetVirtualMemorySize() - 1);
|
||||
LoadConstant(x3, Thread->LookupCache->GetVirtualMemorySize() - 1);
|
||||
and_(x3, RipReg, x3);
|
||||
|
||||
{
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it BlockCache.h::FindBlock
|
||||
LoadConstant(x0, Thread->BlockCache->GetPagePointer());
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
LoadConstant(x0, Thread->LookupCache->GetPagePointer());
|
||||
|
||||
// Offset the address and add to our page pointer
|
||||
lsr(x1, x3, 12);
|
||||
@@ -220,16 +220,16 @@ DispatchGenerator::DispatchGenerator(FEXCore::Context::Context *ctx, FEXCore::Co
|
||||
and_(x1, x3, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(x0, x0, Operand(x1, Shift::LSL, (int)log2(sizeof(FEXCore::BlockCache::BlockCacheEntry))));
|
||||
add(x0, x0, Operand(x1, Shift::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry))));
|
||||
|
||||
// Load the guest address first to ensure it maps to the address we are currently at
|
||||
// This fixes aliasing problems
|
||||
ldr(x1, MemOperand(x0, offsetof(FEXCore::BlockCache::BlockCacheEntry, GuestCode)));
|
||||
ldr(x1, MemOperand(x0, offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)));
|
||||
cmp(x1, RipReg);
|
||||
b(&NoBlock, Condition::ne);
|
||||
|
||||
// Now load the actual host block to execute if we can
|
||||
ldr(x1, MemOperand(x0, offsetof(FEXCore::BlockCache::BlockCacheEntry, HostCode)));
|
||||
ldr(x1, MemOperand(x0, offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)));
|
||||
cbz(x1, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
@@ -540,6 +540,10 @@ void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore:
|
||||
Generator = new DispatchGenerator(ctx, Thread);
|
||||
DispatchPtr = Generator->DispatchPtr;
|
||||
CallbackPtr = Generator->CallbackPtr;
|
||||
|
||||
// TODO: Implement this. It is missing from the dispatcher
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = nullptr;
|
||||
}
|
||||
|
||||
bool InterpreterCore::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/BlockCache.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
@@ -22,19 +22,16 @@ public:
|
||||
explicit InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
~InterpreterCore() override;
|
||||
std::string GetName() override { return "Interpreter"; }
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData) override;
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
void ExecuteCode(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
void DeleteAsmDispatch();
|
||||
|
||||
using CallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
CallbackReturn ReturnPtr;
|
||||
bool HandleSIGBUS(int Signal, void *info, void *ucontext);
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
|
||||
+50
-4618
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,19 @@
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
template<bool copy>
|
||||
class IRListView;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core{
|
||||
struct DebugData;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class InterpreterOps {
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEXCore::IR::IRListView<true> *CurrentIR, FEXCore::Core::DebugData *DebugData);
|
||||
};
|
||||
};
|
||||
@@ -1,5 +1,6 @@
|
||||
#include "Common/MathUtils.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
|
||||
#include <cmath>
|
||||
@@ -17,7 +18,7 @@ class DispatchGenerator : public Xbyak::CodeGenerator {
|
||||
|
||||
CPUBackend::AsmDispatch DispatchPtr;
|
||||
CPUBackend::JITCallback CallbackPtr;
|
||||
InterpreterCore::CallbackReturn ReturnPtr;
|
||||
FEXCore::Context::Context::IntCallbackReturn ReturnPtr;
|
||||
|
||||
uint64_t ThreadStopHandlerAddress;
|
||||
uint64_t AbsoluteLoopTopAddress;
|
||||
@@ -94,13 +95,13 @@ DispatchGenerator::DispatchGenerator(FEXCore::Context::Context *ctx, FEXCore::Co
|
||||
AbsoluteLoopTopAddress = getCurr<uint64_t>();
|
||||
|
||||
{
|
||||
mov(r13, Thread->BlockCache->GetPagePointer());
|
||||
mov(r13, Thread->LookupCache->GetPagePointer());
|
||||
|
||||
// Load our RIP
|
||||
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
|
||||
|
||||
mov(rax, rdx);
|
||||
mov(rbx, Thread->BlockCache->GetVirtualMemorySize() - 1);
|
||||
mov(rbx, Thread->LookupCache->GetVirtualMemorySize() - 1);
|
||||
and_(rax, rbx);
|
||||
shr(rax, 12);
|
||||
|
||||
@@ -113,7 +114,7 @@ DispatchGenerator::DispatchGenerator(FEXCore::Context::Context *ctx, FEXCore::Co
|
||||
mov (rax, rdx);
|
||||
and_(rax, 0x0FFF);
|
||||
|
||||
shl(rax, (int)log2(sizeof(FEXCore::BlockCache::BlockCacheEntry)));
|
||||
shl(rax, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
|
||||
// check for aliasing
|
||||
mov(rcx, qword [rdi + rax + 8]);
|
||||
@@ -238,7 +239,7 @@ DispatchGenerator::DispatchGenerator(FEXCore::Context::Context *ctx, FEXCore::Co
|
||||
|
||||
|
||||
{
|
||||
ReturnPtr = getCurr<InterpreterCore::CallbackReturn>();
|
||||
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
|
||||
// using CallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
|
||||
// rdi = thread
|
||||
@@ -447,7 +448,9 @@ void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore:
|
||||
Generator = new DispatchGenerator(ctx, Thread);
|
||||
DispatchPtr = Generator->DispatchPtr;
|
||||
CallbackPtr = Generator->CallbackPtr;
|
||||
ReturnPtr = Generator->ReturnPtr;
|
||||
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = Generator->ReturnPtr;
|
||||
}
|
||||
|
||||
bool InterpreterCore::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
|
||||
@@ -84,9 +84,9 @@ DEF_OP(ExitFunction) {
|
||||
RipReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
// L1 Cache
|
||||
LoadConstant(x0, State->BlockCache->GetL1Pointer());
|
||||
LoadConstant(x0, State->LookupCache->GetL1Pointer());
|
||||
|
||||
and_(x3, RipReg, BlockCache::L1_ENTRIES_MASK);
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
|
||||
ldp(x1, x0, MemOperand(x0));
|
||||
@@ -260,7 +260,7 @@ DEF_OP(Thunk) {
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
uint8_t *NewCode = (uint8_t *)Op->CodePtr;
|
||||
uint8_t *OldCode = (uint8_t *)&Op->CodeOriginal;
|
||||
uint8_t *OldCode = (uint8_t *)&Op->CodeOriginalLow;
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
|
||||
+83
-94
@@ -1,5 +1,6 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
|
||||
@@ -7,6 +8,7 @@
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <sys/mman.h>
|
||||
#include <stdio.h>
|
||||
@@ -395,6 +397,8 @@ bool JITCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
_mcontext->pc -= 4;
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
uint32_t STR = 0b0011'1000'0011'1111'0110'1000'0000'0000;
|
||||
@@ -404,15 +408,37 @@ bool JITCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STR;
|
||||
PC[1] = DMB;
|
||||
|
||||
// Back up one instruction and have another go
|
||||
_mcontext->pc -= 4;
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(_mcontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
_mcontext->pc += 4;
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS CASPAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(_mcontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
_mcontext->pc += 4;
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS CASAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
|
||||
// Back up one instruction and have another go
|
||||
_mcontext->pc -= 4;
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(&PC[-1], 16);
|
||||
return true;
|
||||
}
|
||||
@@ -658,27 +684,21 @@ void JITCore::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
}
|
||||
}
|
||||
|
||||
struct PhysReg { uint32_t Class; uint32_t VId; };
|
||||
static IR::PhysicalRegister GetPhys(IR::RegisterAllocationData *RAData, uint32_t Node) {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
static PhysReg GetPhys(IR::RegisterAllocationPass *RAPass, uint32_t Node) {
|
||||
uint64_t Reg = RAPass->GetNodeRegister(Node);
|
||||
auto rv = PhysReg {uint32_t(Reg>>32), (uint32_t)Reg};
|
||||
LogMan::Throw::A(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa%d. Class: %d", Node, PhyReg.Class);
|
||||
|
||||
if (rv.VId != ~0U)
|
||||
return rv;
|
||||
else
|
||||
LogMan::Msg::A("Couldn't Allocate register for node: ssa%d. Class: %d", Node, Reg >> 32);
|
||||
|
||||
return PhysReg { ~0U, ~0U};
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
template<>
|
||||
aarch64::Register JITCore::GetReg<JITCore::RA_32>(uint32_t Node) {
|
||||
auto Reg = GetPhys(RAPass, Node);
|
||||
auto Reg = GetPhys(RAData, Node);
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.VId].W();
|
||||
return SRA64[Reg.Reg].W();
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.VId].W();
|
||||
return RA64[Reg.Reg].W();
|
||||
} else {
|
||||
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
@@ -686,11 +706,11 @@ auto Reg = GetPhys(RAPass, Node);
|
||||
|
||||
template<>
|
||||
aarch64::Register JITCore::GetReg<JITCore::RA_64>(uint32_t Node) {
|
||||
auto Reg = GetPhys(RAPass, Node);
|
||||
auto Reg = GetPhys(RAData, Node);
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.VId];
|
||||
return SRA64[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.VId];
|
||||
return RA64[Reg.Reg];
|
||||
} else {
|
||||
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
@@ -698,33 +718,33 @@ aarch64::Register JITCore::GetReg<JITCore::RA_64>(uint32_t Node) {
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> JITCore::GetSrcPair<JITCore::RA_32>(uint32_t Node) {
|
||||
uint32_t Reg = GetPhys(RAPass, Node).VId;
|
||||
uint32_t Reg = GetPhys(RAData, Node).Reg;
|
||||
return RA32Pair[Reg];
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> JITCore::GetSrcPair<JITCore::RA_64>(uint32_t Node) {
|
||||
uint32_t Reg = GetPhys(RAPass, Node).VId;
|
||||
uint32_t Reg = GetPhys(RAData, Node).Reg;
|
||||
return RA64Pair[Reg];
|
||||
}
|
||||
|
||||
aarch64::VRegister JITCore::GetSrc(uint32_t Node) {
|
||||
auto Reg = GetPhys(RAPass, Node);
|
||||
auto Reg = GetPhys(RAData, Node);
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.VId];
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.VId];
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
}
|
||||
|
||||
aarch64::VRegister JITCore::GetDst(uint32_t Node) {
|
||||
auto Reg = GetPhys(RAPass, Node);
|
||||
auto Reg = GetPhys(RAData, Node);
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.VId];
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.VId];
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
@@ -745,8 +765,7 @@ bool JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Va
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType JITCore::GetRegClass(uint32_t Node) {
|
||||
auto Class = static_cast<uint32_t>(RAPass->GetNodeRegister(Node) >> 32);
|
||||
return FEXCore::IR::RegisterClassType {Class};
|
||||
return FEXCore::IR::RegisterClassType {GetPhys(RAData, Node).Class};
|
||||
}
|
||||
|
||||
|
||||
@@ -762,11 +781,13 @@ bool JITCore::IsGPR(uint32_t Node) {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData) {
|
||||
void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
using namespace aarch64;
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->RAData = RAData;
|
||||
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
#ifndef NDEBUG
|
||||
@@ -778,7 +799,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
State->CTX->ClearCodeCache(State, HeaderOp->Entry);
|
||||
State->CTX->ClearCodeCache(State, false);
|
||||
}
|
||||
|
||||
// AAPCS64
|
||||
@@ -834,11 +855,17 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::ThreadState, State.rip)));
|
||||
|
||||
LoadConstant(x0, ThreadSharedData.InterpreterFallbackHelperAddress);
|
||||
LoadConstant(x1, (uintptr_t)IR);
|
||||
|
||||
// Debug data is only used in debug builds
|
||||
#ifndef NDEBUG
|
||||
LoadConstant(x2, (uintptr_t)DebugData);
|
||||
#endif
|
||||
br(x0);
|
||||
} else {
|
||||
LogMan::Throw::A(RAPass->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
//LogMan::Throw::A(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
|
||||
SpillSlots = RAPass->SpillSlots();
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
sub(sp, sp, SpillSlots * 16);
|
||||
@@ -991,7 +1018,7 @@ void JITCore::PopCalleeSavedRegisters() {
|
||||
uint64_t JITCore::ExitFunctionLink(JITCore *core, FEXCore::Core::InternalThreadState *Thread, uint64_t *record) {
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->BlockCache->FindBlock(GuestRip);
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//printf("ExitFunctionLink: Aborting, %lX not in cache\n", GuestRip);
|
||||
@@ -1007,13 +1034,15 @@ uint64_t JITCore::ExitFunctionLink(JITCore *core, FEXCore::Core::InternalThreadS
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->BlockCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
@@ -1026,7 +1055,7 @@ uint64_t JITCore::ExitFunctionLink(JITCore *core, FEXCore::Core::InternalThreadS
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->BlockCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
@@ -1066,15 +1095,14 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// }
|
||||
|
||||
|
||||
uint64_t VirtualMemorySize = Thread->BlockCache->GetVirtualMemorySize();
|
||||
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
|
||||
Literal l_VirtualMemory {VirtualMemorySize};
|
||||
Literal l_PagePtr {Thread->BlockCache->GetPagePointer()};
|
||||
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
|
||||
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
|
||||
Literal l_Interpreter {reinterpret_cast<uint64_t>(State->IntBackend->CompileCode(nullptr, nullptr))};
|
||||
Literal l_Interpreter {reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR)};
|
||||
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
|
||||
|
||||
uintptr_t CompileBlockPtr{};
|
||||
uintptr_t CompileFallbackPtr{};
|
||||
{
|
||||
using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::InternalThreadState *, uint64_t);
|
||||
union PtrCast {
|
||||
@@ -1086,20 +1114,8 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Ptr.ClassPtr = &FEXCore::Context::Context::CompileBlock;
|
||||
CompileBlockPtr = Ptr.Data;
|
||||
}
|
||||
{
|
||||
using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::InternalThreadState *, uint64_t);
|
||||
union PtrCast {
|
||||
ClassPtrType ClassPtr;
|
||||
uintptr_t Data;
|
||||
};
|
||||
|
||||
PtrCast Ptr;
|
||||
Ptr.ClassPtr = &FEXCore::Context::Context::CompileFallbackBlock;
|
||||
CompileFallbackPtr = Ptr.Data;
|
||||
}
|
||||
|
||||
Literal l_CompileBlock {CompileBlockPtr};
|
||||
Literal l_CompileFallback {CompileFallbackPtr};
|
||||
Literal l_ExitFunctionLink {(uintptr_t)&ExitFunctionLink};
|
||||
|
||||
// Push all the register we need to save
|
||||
@@ -1123,9 +1139,7 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
};
|
||||
|
||||
// used from signals
|
||||
aarch64::Label LoopTopFillSRA{};
|
||||
bind(&LoopTopFillSRA);
|
||||
AbsoluteLoopTopAddressFillSRA = GetLabelAddress<uint64_t>(&LoopTopFillSRA);
|
||||
AbsoluteLoopTopAddressFillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
@@ -1134,8 +1148,6 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
aarch64::Label FullLookup{};
|
||||
aarch64::Label LoopTop{};
|
||||
aarch64::Label ExitSpillSRA{};
|
||||
aarch64::Label ThreadPauseHandlerSpillSRA{};
|
||||
aarch64::Label ExitFunctionLinker{};
|
||||
|
||||
bind(&LoopTop);
|
||||
AbsoluteLoopTopAddress = GetLabelAddress<uint64_t>(&LoopTop);
|
||||
@@ -1146,9 +1158,9 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
auto RipReg = x2;
|
||||
|
||||
// L1 Cache
|
||||
LoadConstant(x0, Thread->BlockCache->GetL1Pointer());
|
||||
LoadConstant(x0, Thread->LookupCache->GetL1Pointer());
|
||||
|
||||
and_(x3, RipReg, BlockCache::L1_ENTRIES_MASK);
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
ldp(x1, x0, MemOperand(x0));
|
||||
cmp(x0, RipReg);
|
||||
@@ -1159,12 +1171,12 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
bind(&FullLookup);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it BlockCache.h::FindBlock
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(x0, &l_PagePtr);
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
if (__builtin_popcountl(VirtualMemorySize) == 1) {
|
||||
and_(x3, RipReg, Thread->BlockCache->GetVirtualMemorySize() - 1);
|
||||
and_(x3, RipReg, Thread->LookupCache->GetVirtualMemorySize() - 1);
|
||||
}
|
||||
else {
|
||||
ldr(x3, &l_VirtualMemory);
|
||||
@@ -1186,24 +1198,24 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
and_(x1, x3, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(x0, x0, Operand(x1, Shift::LSL, (int)log2(sizeof(FEXCore::BlockCache::BlockCacheEntry))));
|
||||
add(x0, x0, Operand(x1, Shift::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry))));
|
||||
|
||||
// Load the guest address first to ensure it maps to the address we are currently at
|
||||
// This fixes aliasing problems
|
||||
ldr(x1, MemOperand(x0, offsetof(FEXCore::BlockCache::BlockCacheEntry, GuestCode)));
|
||||
ldr(x1, MemOperand(x0, offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)));
|
||||
cmp(x1, RipReg);
|
||||
b(&NoBlock, Condition::ne);
|
||||
|
||||
// Now load the actual host block to execute if we can
|
||||
ldr(x3, MemOperand(x0, offsetof(FEXCore::BlockCache::BlockCacheEntry, HostCode)));
|
||||
ldr(x3, MemOperand(x0, offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)));
|
||||
cbz(x3, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
LoadConstant(x0, Thread->BlockCache->GetL1Pointer());
|
||||
LoadConstant(x0, Thread->LookupCache->GetL1Pointer());
|
||||
|
||||
and_(x1, RipReg, BlockCache::L1_ENTRIES_MASK);
|
||||
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x1, Shift::LSL, 4));
|
||||
stp(x3, x2, MemOperand(x0));
|
||||
br(x3);
|
||||
@@ -1211,7 +1223,7 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
}
|
||||
|
||||
{
|
||||
b(&ExitSpillSRA);
|
||||
bind(&ExitSpillSRA);
|
||||
ThreadStopHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
SpillStaticRegs();
|
||||
ThreadStopHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
@@ -1224,8 +1236,7 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
}
|
||||
|
||||
{
|
||||
bind(&ExitFunctionLinker);
|
||||
ExitFunctionLinkerAddress = GetLabelAddress<uint64_t>(&ExitFunctionLinker);
|
||||
ExitFunctionLinkerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
|
||||
SpillStaticRegs();
|
||||
|
||||
@@ -1240,7 +1251,6 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
br(x0);
|
||||
}
|
||||
|
||||
aarch64::Label FallbackCore;
|
||||
// Need to create the block
|
||||
{
|
||||
bind(&NoBlock);
|
||||
@@ -1254,31 +1264,12 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
blr(x3); // { CTX, ThreadState, RIP}
|
||||
FillStaticRegs();
|
||||
|
||||
// X0 now contains either nullptr or block pointer
|
||||
cbz(x0, &FallbackCore);
|
||||
// X0 now contains the block pointer
|
||||
blr(x0);
|
||||
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
// We need to fallback to our fallback core
|
||||
{
|
||||
bind(&FallbackCore);
|
||||
|
||||
ldr(x0, &l_CTX);
|
||||
mov(x1, STATE);
|
||||
ldr(x3, &l_CompileFallback);
|
||||
|
||||
// X2 contains our guest RIP
|
||||
SpillStaticRegs();
|
||||
blr(x3); // {ThreadState, RIP}
|
||||
FillStaticRegs();
|
||||
|
||||
// X0 now contains either nullptr or block pointer
|
||||
cbz(x0, &ExitSpillSRA);
|
||||
br(x0);
|
||||
}
|
||||
|
||||
{
|
||||
Label RestoreContextStateHelperLabel{};
|
||||
bind(&RestoreContextStateHelperLabel);
|
||||
@@ -1295,15 +1286,14 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
ThreadSharedData.InterpreterFallbackHelperAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
SpillStaticRegs();
|
||||
mov(x0, STATE);
|
||||
ldr(x1, &l_Interpreter);
|
||||
ldr(x3, &l_Interpreter);
|
||||
|
||||
blr(x1);
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
bind(&ThreadPauseHandlerSpillSRA);
|
||||
ThreadPauseHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
SpillStaticRegs();
|
||||
ThreadPauseHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
@@ -1380,7 +1370,6 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
place(&l_Interpreter);
|
||||
place(&l_Sleep);
|
||||
place(&l_CompileBlock);
|
||||
place(&l_CompileFallback);
|
||||
place(&l_ExitFunctionLink);
|
||||
|
||||
FinalizeCode();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/BlockCache.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
@@ -78,7 +78,7 @@ public:
|
||||
|
||||
~JITCore() override;
|
||||
std::string GetName() override { return "JIT"; }
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData) override;
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -232,6 +232,7 @@ private:
|
||||
|
||||
CompilerSharedData ThreadSharedData;
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
|
||||
|
||||
@@ -909,7 +909,17 @@ DEF_OP(VZip2) {
|
||||
}
|
||||
|
||||
DEF_OP(VBSL) {
|
||||
LogMan::Msg::A("Unimplemented");
|
||||
auto Op = IROp->C<IR::IROp_VBSL>();
|
||||
if (IROp->Size == 16) {
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
bsl(VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B(), GetSrc(Op->Header.Args[2].ID()).V16B());
|
||||
mov(GetDst(Node).V16B(), VTMP1.V16B());
|
||||
}
|
||||
else {
|
||||
mov(VTMP1.V8B(), GetSrc(Op->Header.Args[0].ID()).V8B());
|
||||
bsl(VTMP1.V8B(), GetSrc(Op->Header.Args[1].ID()).V8B(), GetSrc(Op->Header.Args[2].ID()).V8B());
|
||||
mov(GetDst(Node).V8B(), VTMP1.V8B());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VCMPEQ) {
|
||||
|
||||
@@ -81,10 +81,10 @@ DEF_OP(ExitFunction) {
|
||||
Xbyak::Reg RipReg = GetSrc<RA_64>(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
mov(rcx, ThreadState->BlockCache->GetL1Pointer());
|
||||
mov(rcx, ThreadState->LookupCache->GetL1Pointer());
|
||||
mov(rax, RipReg);
|
||||
|
||||
and_(rax, BlockCache::L1_ENTRIES_MASK);
|
||||
and_(rax, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rax, 4);
|
||||
|
||||
Xbyak::RegExp LookupBase = rcx + rax;
|
||||
@@ -239,7 +239,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
uint8_t* OldCode = (uint8_t*)&Op->CodeOriginal;
|
||||
uint8_t* OldCode = (uint8_t*)&Op->CodeOriginalLow;
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
|
||||
+58
-53
@@ -11,6 +11,8 @@
|
||||
#include <cmath>
|
||||
#include <signal.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
// #define DEBUG_RA 1
|
||||
// #define DEBUG_CYCLES
|
||||
|
||||
@@ -460,27 +462,20 @@ void JITCore::ClearCache() {
|
||||
}
|
||||
}
|
||||
|
||||
uint32_t JITCore::GetPhys(uint32_t Node) {
|
||||
uint64_t Reg = RAPass->GetNodeRegister(Node);
|
||||
IR::PhysicalRegister JITCore::GetPhys(uint32_t Node) {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
if ((uint32_t)Reg != ~0U)
|
||||
return Reg;
|
||||
else
|
||||
LogMan::Msg::A("Couldn't Allocate register for node: ssa%d. Class: %d", Node, Reg >> 32);
|
||||
LogMan::Throw::A(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa%d. Class: %d", Node, PhyReg.Class);
|
||||
|
||||
return ~0U;
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
bool JITCore::IsFPR(uint32_t Node) {
|
||||
auto Class = RAPass->GetNodeRegister(Node) >> 32;
|
||||
|
||||
return Class == IR::FPRClass.Val;
|
||||
return RAData->GetNodeRegister(Node).Class == IR::FPRClass.Val;
|
||||
}
|
||||
|
||||
bool JITCore::IsGPR(uint32_t Node) {
|
||||
auto Class = RAPass->GetNodeRegister(Node) >> 32;
|
||||
|
||||
return Class == IR::GPRClass.Val;
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRClass.Val;
|
||||
}
|
||||
|
||||
template<uint8_t RAType>
|
||||
@@ -489,17 +484,17 @@ Xbyak::Reg JITCore::GetSrc(uint32_t Node) {
|
||||
// r10
|
||||
// Callee Saved
|
||||
// rbx, rbp, r12, r13, r14, r15
|
||||
uint32_t Reg = GetPhys(Node);
|
||||
auto PhyReg = GetPhys(Node);
|
||||
if (RAType == RA_64)
|
||||
return RA64[Reg].cvt64();
|
||||
return RA64[PhyReg.Reg].cvt64();
|
||||
else if (RAType == RA_XMM)
|
||||
return RAXMM[Reg];
|
||||
return RAXMM[PhyReg.Reg];
|
||||
else if (RAType == RA_32)
|
||||
return RA64[Reg].cvt32();
|
||||
return RA64[PhyReg.Reg].cvt32();
|
||||
else if (RAType == RA_16)
|
||||
return RA64[Reg].cvt16();
|
||||
return RA64[PhyReg.Reg].cvt16();
|
||||
else if (RAType == RA_8)
|
||||
return RA64[Reg].cvt8();
|
||||
return RA64[PhyReg.Reg].cvt8();
|
||||
}
|
||||
|
||||
template
|
||||
@@ -515,23 +510,23 @@ template
|
||||
Xbyak::Reg JITCore::GetSrc<JITCore::RA_8>(uint32_t Node);
|
||||
|
||||
Xbyak::Xmm JITCore::GetSrc(uint32_t Node) {
|
||||
uint32_t Reg = GetPhys(Node);
|
||||
return RAXMM_x[Reg];
|
||||
auto PhyReg = GetPhys(Node);
|
||||
return RAXMM_x[PhyReg.Reg];
|
||||
}
|
||||
|
||||
template<uint8_t RAType>
|
||||
Xbyak::Reg JITCore::GetDst(uint32_t Node) {
|
||||
uint32_t Reg = GetPhys(Node);
|
||||
auto PhyReg = GetPhys(Node);
|
||||
if (RAType == RA_64)
|
||||
return RA64[Reg].cvt64();
|
||||
return RA64[PhyReg.Reg].cvt64();
|
||||
else if (RAType == RA_XMM)
|
||||
return RAXMM[Reg];
|
||||
return RAXMM[PhyReg.Reg];
|
||||
else if (RAType == RA_32)
|
||||
return RA64[Reg].cvt32();
|
||||
return RA64[PhyReg.Reg].cvt32();
|
||||
else if (RAType == RA_16)
|
||||
return RA64[Reg].cvt16();
|
||||
return RA64[PhyReg.Reg].cvt16();
|
||||
else if (RAType == RA_8)
|
||||
return RA64[Reg].cvt8();
|
||||
return RA64[PhyReg.Reg].cvt8();
|
||||
}
|
||||
|
||||
template
|
||||
@@ -548,11 +543,11 @@ Xbyak::Reg JITCore::GetDst<JITCore::RA_8>(uint32_t Node);
|
||||
|
||||
template<uint8_t RAType>
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> JITCore::GetSrcPair(uint32_t Node) {
|
||||
uint32_t Reg = GetPhys(Node);
|
||||
auto PhyReg = GetPhys(Node);
|
||||
if (RAType == RA_64)
|
||||
return RA64Pair[Reg];
|
||||
return RA64Pair[PhyReg.Reg];
|
||||
else if (RAType == RA_32)
|
||||
return {RA64Pair[Reg].first.cvt32(), RA64Pair[Reg].second.cvt32()};
|
||||
return {RA64Pair[PhyReg.Reg].first.cvt32(), RA64Pair[PhyReg.Reg].second.cvt32()};
|
||||
}
|
||||
|
||||
template
|
||||
@@ -562,8 +557,8 @@ template
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> JITCore::GetSrcPair<JITCore::RA_32>(uint32_t Node);
|
||||
|
||||
Xbyak::Xmm JITCore::GetDst(uint32_t Node) {
|
||||
uint32_t Reg = GetPhys(Node);
|
||||
return RAXMM_x[Reg];
|
||||
auto PhyReg = GetPhys(Node);
|
||||
return RAXMM_x[PhyReg.Reg];
|
||||
}
|
||||
|
||||
bool JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
|
||||
@@ -587,7 +582,7 @@ std::tuple<JITCore::SetCC, JITCore::CMovCC, JITCore::JCC> JITCore::GetCC(IR::Con
|
||||
case FEXCore::IR::COND_SGE: return { &CodeGenerator::setge, &CodeGenerator::cmovge, &CodeGenerator::jge };
|
||||
case FEXCore::IR::COND_SLT: return { &CodeGenerator::setl , &CodeGenerator::cmovl , &CodeGenerator::jl };
|
||||
case FEXCore::IR::COND_SGT: return { &CodeGenerator::setg , &CodeGenerator::cmovg , &CodeGenerator::jg };
|
||||
case FEXCore::IR::COND_SLE: return { &CodeGenerator::setle, &CodeGenerator::cmovle, &CodeGenerator::jle };
|
||||
case FEXCore::IR::COND_SLE: return { &CodeGenerator::setle, &CodeGenerator::cmovle, &CodeGenerator::jle };
|
||||
case FEXCore::IR::COND_UGE: return { &CodeGenerator::setae, &CodeGenerator::cmovae, &CodeGenerator::jae };
|
||||
case FEXCore::IR::COND_ULT: return { &CodeGenerator::setb , &CodeGenerator::cmovb , &CodeGenerator::jb };
|
||||
case FEXCore::IR::COND_UGT: return { &CodeGenerator::seta , &CodeGenerator::cmova , &CodeGenerator::ja };
|
||||
@@ -608,18 +603,22 @@ std::tuple<JITCore::SetCC, JITCore::CMovCC, JITCore::JCC> JITCore::GetCC(IR::Con
|
||||
LogMan::Msg::A("Unsupported compare type");
|
||||
break;
|
||||
}
|
||||
|
||||
// Hope for the best
|
||||
return { &CodeGenerator::sete , &CodeGenerator::cmove , &CodeGenerator::je };
|
||||
}
|
||||
|
||||
void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData) {
|
||||
void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->RAData = RAData;
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
if ((getSize() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState, HeaderOp->Entry);
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState, false);
|
||||
}
|
||||
|
||||
void *Entry = getCurr<void*>();
|
||||
@@ -647,12 +646,19 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const
|
||||
if (HeaderOp->ShouldInterpret) {
|
||||
mov(rax, HeaderOp->Entry);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CPUState, rip)], rax);
|
||||
mov(rsi, (uint64_t)IR);
|
||||
|
||||
// Debug data is only used in debug builds
|
||||
#ifndef NDEBUG
|
||||
mov(rdx, (uint64_t)DebugData);
|
||||
#endif
|
||||
|
||||
mov(rax, (uintptr_t)ThreadSharedData.InterpreterFallbackHelperAddress);
|
||||
jmp(rax);
|
||||
} else {
|
||||
LogMan::Throw::A(RAPass->HasFullRA(), "Needs RA");
|
||||
LogMan::Throw::A(RAData != nullptr, "Needs RA");
|
||||
|
||||
SpillSlots = RAPass->SpillSlots();
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
sub(rsp, SpillSlots * 16);
|
||||
@@ -777,7 +783,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
}
|
||||
|
||||
|
||||
void *Exit = getCurr<void*>();
|
||||
this->IR = nullptr;
|
||||
|
||||
@@ -804,7 +810,7 @@ static void SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::InternalT
|
||||
uint64_t JITCore::ExitFunctionLink(JITCore *core, FEXCore::Core::InternalThreadState *Thread, uint64_t *record) {
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->BlockCache->FindBlock(GuestRip);
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
Thread->State.State.rip = GuestRip;
|
||||
@@ -812,7 +818,7 @@ uint64_t JITCore::ExitFunctionLink(JITCore *core, FEXCore::Core::InternalThreadS
|
||||
}
|
||||
|
||||
auto LinkerAddress = core->ExitFunctionLinkerAddress;
|
||||
Thread->BlockCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
@@ -884,21 +890,21 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
|
||||
|
||||
// L1 Cache
|
||||
mov(r13, Thread->BlockCache->GetL1Pointer());
|
||||
mov(r13, Thread->LookupCache->GetL1Pointer());
|
||||
mov(rax, rdx);
|
||||
|
||||
and_(rax, BlockCache::L1_ENTRIES_MASK);
|
||||
and_(rax, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rax, 4);
|
||||
cmp(qword[r13 + rax + 8], rdx);
|
||||
jne(FullLookup);
|
||||
jmp(qword[r13 + rax + 0]);
|
||||
|
||||
L(FullLookup);
|
||||
mov(r13, Thread->BlockCache->GetPagePointer());
|
||||
|
||||
L(FullLookup);
|
||||
mov(r13, Thread->LookupCache->GetPagePointer());
|
||||
|
||||
// Full lookup
|
||||
mov(rax, rdx);
|
||||
mov(rbx, Thread->BlockCache->GetVirtualMemorySize() - 1);
|
||||
mov(rbx, Thread->LookupCache->GetVirtualMemorySize() - 1);
|
||||
and_(rax, rbx);
|
||||
shr(rax, 12);
|
||||
|
||||
@@ -911,7 +917,7 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
mov (rax, rdx);
|
||||
and(rax, 0x0FFF);
|
||||
|
||||
shl(rax, (int)log2(sizeof(FEXCore::BlockCache::BlockCacheEntry)));
|
||||
shl(rax, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
|
||||
// check for aliasing
|
||||
mov(rcx, qword [rdi + rax + 8]);
|
||||
@@ -925,13 +931,13 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
je(NoBlock);
|
||||
|
||||
// Update L1
|
||||
mov(r13, Thread->BlockCache->GetL1Pointer());
|
||||
mov(r13, Thread->LookupCache->GetL1Pointer());
|
||||
mov(rcx, rdx);
|
||||
and_(rcx, BlockCache::L1_ENTRIES_MASK);
|
||||
and_(rcx, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rcx, 1);
|
||||
mov(qword[r13 + rcx*8 + 8], rdx);
|
||||
mov(qword[r13 + rcx*8 + 0], rax);
|
||||
|
||||
|
||||
// Real block if we made it here
|
||||
jmp(rax);
|
||||
}
|
||||
@@ -962,7 +968,7 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
call(rax);
|
||||
jmp(rax);
|
||||
}
|
||||
|
||||
|
||||
Label FallbackCore;
|
||||
// Block creation
|
||||
{
|
||||
@@ -1000,9 +1006,8 @@ void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Interpreter fallback helper code
|
||||
ThreadSharedData.InterpreterFallbackHelperAddress = getCurr<void*>();
|
||||
// This will get called so our stack is now misaligned
|
||||
mov(rax, reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR));
|
||||
mov(rdi, STATE);
|
||||
mov(rax, reinterpret_cast<uint64_t>(ThreadState->IntBackend->CompileCode(nullptr, nullptr)));
|
||||
|
||||
call(rax);
|
||||
|
||||
jmp(LoopTop);
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/BlockCache.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JIT.h"
|
||||
@@ -14,6 +14,7 @@ using namespace Xbyak;
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <tuple>
|
||||
|
||||
@@ -58,7 +59,7 @@ public:
|
||||
explicit JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer, bool CompileThread);
|
||||
~JITCore() override;
|
||||
std::string GetName() override { return "JIT"; }
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData) override;
|
||||
void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -106,7 +107,7 @@ private:
|
||||
constexpr static uint8_t RA_64 = 3;
|
||||
constexpr static uint8_t RA_XMM = 4;
|
||||
|
||||
uint32_t GetPhys(uint32_t Node);
|
||||
IR::PhysicalRegister GetPhys(uint32_t Node);
|
||||
|
||||
bool IsFPR(uint32_t Node);
|
||||
bool IsGPR(uint32_t Node);
|
||||
@@ -128,6 +129,7 @@ private:
|
||||
|
||||
void CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread);
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
FEXCore::IR::RegisterAllocationData *RAData;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
bool GetSamplingData {true};
|
||||
|
||||
@@ -327,7 +327,7 @@ DEF_OP(VAddV) {
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 2: {
|
||||
for (int i = Elements; i > 1; i >>= 1) {
|
||||
phaddw(Dest, Src);
|
||||
vphaddw(Dest, Src, Dest);
|
||||
Src = Dest;
|
||||
}
|
||||
pextrw(eax, Dest, 0);
|
||||
@@ -336,7 +336,7 @@ DEF_OP(VAddV) {
|
||||
}
|
||||
case 4: {
|
||||
for (int i = Elements; i > 1; i >>= 1) {
|
||||
phaddd(Dest, Src);
|
||||
vphaddd(Dest, Src, Dest);
|
||||
Src = Dest;
|
||||
}
|
||||
pextrd(eax, Dest, 0);
|
||||
|
||||
+14
-6
@@ -1,10 +1,10 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/BlockCache.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include <sys/mman.h>
|
||||
|
||||
namespace FEXCore {
|
||||
BlockCache::BlockCache(FEXCore::Context::Context *CTX)
|
||||
LookupCache::LookupCache(FEXCore::Context::Context *CTX)
|
||||
: ctx {CTX} {
|
||||
|
||||
// Block cache ends up looking like this
|
||||
@@ -36,13 +36,13 @@ BlockCache::BlockCache(FEXCore::Context::Context *CTX)
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
}
|
||||
|
||||
BlockCache::~BlockCache() {
|
||||
LookupCache::~LookupCache() {
|
||||
munmap(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8);
|
||||
munmap(reinterpret_cast<void*>(PageMemory), CODE_SIZE);
|
||||
munmap(reinterpret_cast<void*>(L1Pointer), L1_SIZE);
|
||||
}
|
||||
|
||||
void BlockCache::HintUsedRange(uint64_t Address, uint64_t Size) {
|
||||
void LookupCache::HintUsedRange(uint64_t Address, uint64_t Size) {
|
||||
// Tell the kernel we will definitely need [Address, Address+Size) mapped for the page pointer
|
||||
// Page Pointer is allocated per page, so shift by page size
|
||||
Address >>= 12;
|
||||
@@ -50,14 +50,22 @@ void BlockCache::HintUsedRange(uint64_t Address, uint64_t Size) {
|
||||
madvise(reinterpret_cast<void*>(PagePointer + Address), Size, MADV_WILLNEED);
|
||||
}
|
||||
|
||||
void BlockCache::ClearCache() {
|
||||
void LookupCache::ClearL2Cache() {
|
||||
// Clear out the page memory
|
||||
madvise(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8, MADV_DONTNEED);
|
||||
madvise(reinterpret_cast<void*>(PageMemory), CODE_SIZE, MADV_DONTNEED);
|
||||
madvise(reinterpret_cast<void*>(L1Pointer), L1_SIZE, MADV_DONTNEED);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
// Clear L1
|
||||
madvise(reinterpret_cast<void*>(L1Pointer), L1_SIZE, MADV_DONTNEED);
|
||||
// Clear L2
|
||||
ClearL2Cache();
|
||||
// All code is gone, remove links
|
||||
BlockLinks.clear();
|
||||
// All code is gone, clear the block list
|
||||
BlockList.clear();
|
||||
}
|
||||
|
||||
}
|
||||
+63
-37
@@ -2,23 +2,44 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <map>
|
||||
|
||||
namespace FEXCore {
|
||||
class BlockCache {
|
||||
class LookupCache {
|
||||
public:
|
||||
|
||||
struct BlockCacheEntry {
|
||||
struct LookupCacheEntry {
|
||||
uintptr_t HostCode;
|
||||
uintptr_t GuestCode;
|
||||
};
|
||||
|
||||
BlockCache(FEXCore::Context::Context *CTX);
|
||||
~BlockCache();
|
||||
LookupCache(FEXCore::Context::Context *CTX);
|
||||
~LookupCache();
|
||||
|
||||
using BlockCacheIter = uintptr_t;
|
||||
using LookupCacheIter = uintptr_t;
|
||||
uintptr_t End() { return 0; }
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
return FindCodePointerForAddress(Address);
|
||||
auto HostCode = FindCodePointerForAddress(Address);
|
||||
if (HostCode) {
|
||||
return HostCode;
|
||||
} else {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
|
||||
if (HostCode != BlockList.end()) {
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode) {
|
||||
auto InsertPoint = BlockList.emplace(Address, (uintptr_t)HostCode);
|
||||
LogMan::Throw::A(InsertPoint.second == true, "Dupplicate block mapping added");
|
||||
|
||||
// no need to update L1 or L2, they will get updated on first lookup
|
||||
}
|
||||
|
||||
void Erase(uint64_t Address) {
|
||||
@@ -30,8 +51,11 @@ public:
|
||||
it->second();
|
||||
}
|
||||
|
||||
// Remove from BlockList
|
||||
BlockList.erase(Address);
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<BlockCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = L1Entry.HostCode = 0;
|
||||
}
|
||||
@@ -49,14 +73,32 @@ public:
|
||||
}
|
||||
|
||||
// Page exists, just set the offset to zero
|
||||
auto BlockPointers = reinterpret_cast<BlockCacheEntry*>(LocalPagePointer);
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
BlockPointers[PageOffset].GuestCode = 0;
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
}
|
||||
|
||||
uintptr_t AddBlockMapping(uint64_t Address, void *Ptr) {
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
BlockLinks.insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
|
||||
void HintUsedRange(uint64_t Address, uint64_t Size);
|
||||
|
||||
uintptr_t GetL1Pointer() { return L1Pointer; }
|
||||
uintptr_t GetPagePointer() { return PagePointer; }
|
||||
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
|
||||
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<BlockCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = L1Entry.HostCode = 0;
|
||||
}
|
||||
@@ -74,40 +116,23 @@ public:
|
||||
// Allocate one now if we can
|
||||
uintptr_t NewPageBacking = AllocateBackingForPage();
|
||||
if (!NewPageBacking) {
|
||||
// Couldn't allocate, return so the frontend can recover from this
|
||||
return 0;
|
||||
// Couldn't allocate, clear L2 and retry
|
||||
ClearL2Cache();
|
||||
CacheBlockMapping(Address, HostCode);
|
||||
return;
|
||||
}
|
||||
Pointers[Address] = NewPageBacking;
|
||||
LocalPagePointer = NewPageBacking;
|
||||
}
|
||||
|
||||
// Add the new pointer to the page block
|
||||
auto BlockPointers = reinterpret_cast<BlockCacheEntry*>(LocalPagePointer);
|
||||
uintptr_t CastPtr = reinterpret_cast<uintptr_t>(Ptr);
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
// This silently replaces existing mappings
|
||||
BlockPointers[PageOffset].GuestCode = FullAddress;
|
||||
BlockPointers[PageOffset].HostCode = CastPtr;
|
||||
|
||||
return CastPtr;
|
||||
BlockPointers[PageOffset].HostCode = HostCode;
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
BlockLinks.insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
|
||||
void HintUsedRange(uint64_t Address, uint64_t Size);
|
||||
|
||||
uintptr_t GetL1Pointer() { return L1Pointer; }
|
||||
uintptr_t GetPagePointer() { return PagePointer; }
|
||||
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
|
||||
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
private:
|
||||
uintptr_t AllocateBackingForPage() {
|
||||
uintptr_t NewBase = AllocateOffset;
|
||||
uintptr_t NewEnd = AllocateOffset + SIZE_PER_PAGE;
|
||||
@@ -125,7 +150,7 @@ private:
|
||||
uintptr_t FindCodePointerForAddress(uint64_t Address) {
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<BlockCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
@@ -143,7 +168,7 @@ private:
|
||||
}
|
||||
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<BlockCacheEntry*>(LocalPagePointer);
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == FullAddress)
|
||||
{
|
||||
@@ -173,10 +198,11 @@ private:
|
||||
};
|
||||
|
||||
std::map<BlockLinkTag, std::function<void()>> BlockLinks;
|
||||
std::map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(BlockCacheEntry);
|
||||
constexpr static size_t L1_SIZE = L1_ENTRIES * sizeof(BlockCacheEntry);
|
||||
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t L1_SIZE = L1_ENTRIES * sizeof(LookupCacheEntry);
|
||||
|
||||
size_t AllocateOffset {};
|
||||
|
||||
+18
-34
@@ -206,7 +206,7 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) {
|
||||
//ss
|
||||
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, ss), _LoadMem(GPRClass, GPRSize, SP, GPRSize));
|
||||
SP = _Add(SP, Constant);
|
||||
|
||||
|
||||
_ExitFunction(NewRIP);
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
@@ -637,7 +637,7 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) {
|
||||
_InvalidateFlags(~0UL); // all flags
|
||||
}
|
||||
|
||||
auto ConstantPC = _Constant(Op->PC + Op->InstSize);
|
||||
auto ConstantPC = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
|
||||
OrderedNode *JMPPCOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
@@ -665,7 +665,7 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) {
|
||||
uint8_t Size = GetSrcSize(Op);
|
||||
OrderedNode *JMPPCOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
auto ConstantPCReturn = _Constant(Op->PC + Op->InstSize);
|
||||
auto ConstantPCReturn = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
|
||||
auto ConstantSize = _Constant(Size);
|
||||
auto OldSP = _LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RSP]), GPRClass);
|
||||
@@ -902,7 +902,8 @@ OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, OrderedNode *TrueValue, Ord
|
||||
SrcCond = _Select(FEXCore::IR::COND_FNU, flagsOpDest, flagsOpSrc, TrueValue, FalseValue, flagsOpSize);
|
||||
break;
|
||||
default:
|
||||
printf("Missed Condition %04X FLAGS_OP_FCMP\n", OP); break;
|
||||
// TODO: Add more optimized cases
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -994,9 +995,6 @@ void OpDispatchBuilder::CondJUMPRCXOp(OpcodeArgs) {
|
||||
uint8_t JcxGPRSize = CTX->Config.Is64BitMode ? 8 : 4;
|
||||
JcxGPRSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) ? (JcxGPRSize >> 1) : JcxGPRSize;
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
IRPair<IROp_Header> SrcCond;
|
||||
|
||||
IRPair<IROp_Constant> TakeBranch;
|
||||
IRPair<IROp_Constant> DoNotTakeBranch;
|
||||
TakeBranch = _Constant(1);
|
||||
@@ -1104,7 +1102,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
SetTrueJumpTarget(CondJump, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
|
||||
auto NewRIP = _Constant(Target);
|
||||
auto NewRIP = _Constant(GPRSize * 8, Target);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
@@ -1122,7 +1120,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
|
||||
// Leave block
|
||||
auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
|
||||
auto RIPTargetConst = _Constant(GPRSize * 8, Op->PC + Op->InstSize);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPTargetConst);
|
||||
@@ -1150,7 +1148,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
auto JumpTarget = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetJumpTarget(Jump, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
_ExitFunction(_Constant(Target));
|
||||
_ExitFunction(_Constant(GPRSize * 8, Target));
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -1169,8 +1167,6 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::JUMPAbsoluteOp(OpcodeArgs) {
|
||||
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
|
||||
|
||||
BlockSetRIP = true;
|
||||
// This is just an unconditional jump
|
||||
// This uses ModRM to determine its location
|
||||
@@ -3004,8 +3000,8 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
auto LoopHead = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto LoopTail = CreateNewCodeBlockAfter(LoopHead);
|
||||
auto LoopEnd = CreateNewCodeBlockAfter(LoopTail);
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
// At the time this was written, our RA can't handle accessing nodes across blocks.
|
||||
// So we need to re-load and re-calculate essential values each iteration of the loop.
|
||||
@@ -3022,14 +3018,12 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
auto PtrDir = _Select(FEXCore::IR::COND_EQ,
|
||||
DF, _Constant(0),
|
||||
SizeConst, NegSizeConst);
|
||||
|
||||
|
||||
_Jump(LoopHead);
|
||||
|
||||
SetCurrentCodeBlock(LoopHead);
|
||||
{
|
||||
OrderedNode *Counter = _LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RCX]), GPRClass);
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
// Can we end the block?
|
||||
_CondJump(Counter, LoopEnd, LoopTail, {COND_EQ});
|
||||
}
|
||||
@@ -3087,8 +3081,8 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
auto LoopHead = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto LoopTail = CreateNewCodeBlockAfter(LoopHead);
|
||||
auto LoopEnd = CreateNewCodeBlockAfter(LoopTail);
|
||||
|
||||
|
||||
|
||||
|
||||
// At the time this was written, our RA can't handle accessing nodes across blocks.
|
||||
// So we need to re-load and re-calculate essential values each iteration of the loop.
|
||||
|
||||
@@ -3099,8 +3093,6 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(LoopHead);
|
||||
{
|
||||
OrderedNode *Counter = _LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RCX]), GPRClass);
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
_CondJump(Counter, LoopEnd, LoopTail, {COND_EQ});
|
||||
}
|
||||
|
||||
@@ -3205,7 +3197,7 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
auto PtrDir = _Select(FEXCore::IR::COND_EQ,
|
||||
DF, _Constant(0),
|
||||
_Constant(Size), _Constant(-Size));
|
||||
|
||||
|
||||
auto JumpStart = _Jump();
|
||||
// Make sure to start a new block after ending this one
|
||||
auto LoopStart = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
@@ -3326,9 +3318,6 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
SetJumpTarget(JumpStart, LoopStart);
|
||||
SetCurrentCodeBlock(LoopStart);
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
OrderedNode *Counter = _LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RCX]), GPRClass);
|
||||
|
||||
// Can we end the block?
|
||||
@@ -3418,16 +3407,13 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
auto PtrDir = _Select(FEXCore::IR::COND_EQ,
|
||||
DF, _Constant(0),
|
||||
SizeConst, NegSizeConst);
|
||||
|
||||
|
||||
auto JumpStart = _Jump();
|
||||
// Make sure to start a new block after ending this one
|
||||
auto LoopStart = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetJumpTarget(JumpStart, LoopStart);
|
||||
SetCurrentCodeBlock(LoopStart);
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
OrderedNode *Counter = _LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RCX]), GPRClass);
|
||||
|
||||
// Can we end the block?
|
||||
@@ -3475,9 +3461,9 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
// Make sure to start a new block after ending this one
|
||||
auto LoopEnd = CreateNewCodeBlockAfter(LoopTail);
|
||||
SetTrueJumpTarget(CondJump, LoopEnd);
|
||||
|
||||
|
||||
SetFalseJumpTarget(InternalCondJump, LoopEnd);
|
||||
|
||||
|
||||
SetCurrentCodeBlock(LoopEnd);
|
||||
}
|
||||
}
|
||||
@@ -3979,7 +3965,6 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
|
||||
OrderedNode *CurrentVal = _Constant(0);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
//TODO: We could remove this VCastFromGOR + VInsGPR pair if we had a VDUPFromGPR instruction that maps directly to AArch64.
|
||||
@@ -4442,7 +4427,7 @@ void OpDispatchBuilder::Finalize() {
|
||||
|
||||
// We haven't emitted. Dump out to the dispatcher
|
||||
SetCurrentCodeBlock(Handler.second.BlockEntry);
|
||||
_ExitFunction(_Constant(Handler.first));
|
||||
_ExitFunction(_Constant(GPRSize * 8, Handler.first));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9049,7 +9034,6 @@ constexpr uint16_t PF_F2 = 3;
|
||||
#define OPD(prefix, opcode) ((prefix << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = 1;
|
||||
constexpr uint16_t PF_38_F2 = 2;
|
||||
const std::vector<std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr>> H0F38Table = {
|
||||
{OPD(PF_38_NONE, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
|
||||
{OPD(PF_38_66, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
|
||||
|
||||
+41
-2
@@ -3,11 +3,14 @@
|
||||
#include <cstring>
|
||||
#include <stdlib.h>
|
||||
#include <vector>
|
||||
#include <sys/mman.h>
|
||||
|
||||
namespace FEXCore {
|
||||
constexpr size_t CODE_SIZE = 0x1000;
|
||||
|
||||
X86GeneratedCode::X86GeneratedCode() {
|
||||
// Allocate a page for our emulated guest
|
||||
CodePtr = malloc(0x1000);
|
||||
CodePtr = AllocateGuestCodeSpace(CODE_SIZE);
|
||||
|
||||
SignalReturn = reinterpret_cast<uint64_t>(CodePtr);
|
||||
CallbackReturn = reinterpret_cast<uint64_t>(CodePtr) + 2;
|
||||
@@ -21,7 +24,43 @@ X86GeneratedCode::X86GeneratedCode() {
|
||||
}
|
||||
|
||||
X86GeneratedCode::~X86GeneratedCode() {
|
||||
free(CodePtr);
|
||||
munmap(CodePtr, CODE_SIZE);
|
||||
}
|
||||
|
||||
void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
|
||||
FEXCore::Config::Value<bool> Is64BitMode{FEXCore::Config::CONFIG_IS64BIT_MODE, 0};
|
||||
|
||||
if (Is64BitMode()) {
|
||||
// 64bit mode can have its sigret handler anywhere
|
||||
return mmap(nullptr, Size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
}
|
||||
|
||||
// First 64bit page
|
||||
constexpr uintptr_t LOCATION_MAX = 0x1'0000'0000;
|
||||
|
||||
// 32bit mode
|
||||
// We need to have the sigret handler in the lower 32bits of memory space
|
||||
// Scan top down and try to allocate a location
|
||||
for (size_t Location = 0xFFFF'E000; Location != 0x0; Location -= 0x1000) {
|
||||
void *Ptr = mmap(reinterpret_cast<void*>(Location), Size, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
if (Ptr != MAP_FAILED &&
|
||||
reinterpret_cast<uintptr_t>(Ptr) >= LOCATION_MAX) {
|
||||
// Failed to map in the lower 32bits
|
||||
// Try again
|
||||
// Can happen in the case that host kernel ignores MAP_FIXED_NOREPLACE
|
||||
munmap(Ptr, Size);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Ptr != MAP_FAILED) {
|
||||
return Ptr;
|
||||
}
|
||||
}
|
||||
|
||||
// Can't do anything about this
|
||||
// Here's hoping the application doesn't use signals
|
||||
return MAP_FAILED;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
#pragma once
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -12,5 +14,6 @@ public:
|
||||
|
||||
private:
|
||||
void *CodePtr{};
|
||||
void* AllocateGuestCodeSpace(size_t Size);
|
||||
};
|
||||
}
|
||||
@@ -100,24 +100,6 @@ void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeEVEXTables();
|
||||
|
||||
#ifndef NDEBUG
|
||||
auto CheckTable = [&UnknownOp](auto& FinalTable) {
|
||||
for (size_t i = 0; i < FinalTable.size(); ++i) {
|
||||
auto const &Op = FinalTable.at(i);
|
||||
|
||||
if (Op == UnknownOp) {
|
||||
LogMan::Msg::A("Unknown Op: 0x%lx", i);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// CheckTable(BaseOps);
|
||||
// CheckTable(SecondBaseOps);
|
||||
|
||||
// CheckTable(RepModOps);
|
||||
// CheckTable(RepNEModOps);
|
||||
// CheckTable(OpSizeModOps);
|
||||
// CheckTable(X87Ops);
|
||||
|
||||
X86InstDebugInfo::InstallDebugInfo();
|
||||
LogMan::Msg::D("X86Tables had %ld total insts, and %ld labeled as understood", Total, NumInsts);
|
||||
#endif
|
||||
|
||||
@@ -125,9 +125,12 @@ void InitializePrimaryGroupTables(Context::OperatingMode Mode) {
|
||||
|
||||
// GROUP 11
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 0), 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT, 1, nullptr}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 1), 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 7), 1, X86InstInfo{"XABORT", TYPE_INST, FLAGS_MODRM, 1, nullptr}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 1), 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 7), 1, X86InstInfo{"XBEGIN", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT | FLAGS_SETS_RIP | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
|
||||
};
|
||||
|
||||
const U16U8InfoStruct PrimaryGroupOpTable_64[] = {
|
||||
|
||||
@@ -21,8 +21,8 @@ void InitializeSecondaryModRMTables() {
|
||||
{((1 << 3) | 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((1 << 3) | 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((1 << 3) | 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((1 << 3) | 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((1 << 3) | 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((1 << 3) | 5), 1, X86InstInfo{"XEND", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((1 << 3) | 6), 1, X86InstInfo{"XTEST", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((1 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// REG /3
|
||||
|
||||
+5
-2
@@ -28,7 +28,9 @@
|
||||
"static constexpr FEXCore::IR::RegisterClassType FPRFixedClass {3}",
|
||||
"static constexpr FEXCore::IR::RegisterClassType GPRPairClass {4}",
|
||||
"static constexpr FEXCore::IR::RegisterClassType ComplexClass {5}",
|
||||
"static constexpr FEXCore::IR::RegisterClassType InvalidClass {~0U}",
|
||||
"static constexpr FEXCore::IR::RegisterClassType InvalidClass {7}",
|
||||
"",
|
||||
"static constexpr uint8_t InvalidReg {31}",
|
||||
"",
|
||||
"static const FEXCore::IR::TypeDefinition i8 {TypeDefinition::Create(1, 0)}",
|
||||
"static const FEXCore::IR::TypeDefinition i16 {TypeDefinition::Create(2, 0)}",
|
||||
@@ -129,7 +131,8 @@
|
||||
"DestClass": "GPR",
|
||||
"DestSize": "8",
|
||||
"Args": [
|
||||
"__uint128_t", "CodeOriginal",
|
||||
"uint64_t", "CodeOriginalLow",
|
||||
"uint64_t", "CodeOriginalHigh",
|
||||
"uint64_t", "CodePtr",
|
||||
"uint8_t", "CodeLength"
|
||||
]
|
||||
|
||||
Vendored
+44
-36
@@ -21,7 +21,7 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false>
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false> const* IR, CondClassType Arg) {
|
||||
std::array<std::string, 14> CondNames = {
|
||||
std::array<std::string, 22> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
"UGE",
|
||||
@@ -36,6 +36,14 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false>
|
||||
"SLT",
|
||||
"SGT",
|
||||
"SLE",
|
||||
"Invalid Cond",
|
||||
"Invalid Cond",
|
||||
"FLU",
|
||||
"FGE",
|
||||
"FLEU",
|
||||
"FGT",
|
||||
"FU",
|
||||
"FNU"
|
||||
};
|
||||
|
||||
*out << CondNames[Arg];
|
||||
@@ -66,26 +74,33 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false>
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, IRListView<false> const* IR, OrderedNodeWrapper Arg, IR::RegisterAllocationPass *RAPass) {
|
||||
static void PrintArg(std::stringstream *out, IRListView<false> const* IR, OrderedNodeWrapper Arg, IR::RegisterAllocationData *RAData) {
|
||||
auto [CodeNode, IROp] = IR->at(Arg)();
|
||||
|
||||
*out << "%ssa" << std::to_string(Arg.ID());
|
||||
if (RAPass) {
|
||||
uint64_t RegClass = RAPass->GetNodeRegister(Arg.ID());
|
||||
FEXCore::IR::RegisterClassType Class {uint32_t(RegClass >> 32)};
|
||||
uint32_t Reg = RegClass;
|
||||
switch (Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
}
|
||||
if (Arg.ID() == 0) {
|
||||
*out << "%Invalid";
|
||||
} else {
|
||||
*out << "%ssa" << std::to_string(Arg.ID());
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(Arg.ID());
|
||||
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
}
|
||||
|
||||
*out << std::dec << Reg << ")";
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
|
||||
} else {
|
||||
*out << ")";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (IROp->HasDest) {
|
||||
@@ -108,14 +123,6 @@ static void PrintArg(std::stringstream *out, IRListView<false> const* IR, Ordere
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false> const* IR, IR::TypeDefinition Arg) {
|
||||
*out << "i" << std::dec << static_cast<uint32_t>(Arg.Bytes() * 8);
|
||||
|
||||
if (Arg.Elements()) {
|
||||
*out << "v" << std::dec << static_cast<uint32_t>(Arg.Elements());
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false> const* IR, FEXCore::IR::FenceType Arg) {
|
||||
if (Arg == IR::Fence_Load) {
|
||||
*out << "Loads";
|
||||
@@ -131,7 +138,7 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView<false>
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(std::stringstream *out, IRListView<false> const* IR, IR::RegisterAllocationPass *RAPass) {
|
||||
void Dump(std::stringstream *out, IRListView<false> const* IR, IR::RegisterAllocationData *RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
int8_t CurrentIndent = 0;
|
||||
@@ -188,11 +195,9 @@ void Dump(std::stringstream *out, IRListView<false> const* IR, IR::RegisterAlloc
|
||||
|
||||
*out << "%ssa" << std::to_string(ID);
|
||||
|
||||
if (RAPass) {
|
||||
uint64_t RegClass = RAPass->GetNodeRegister(ID);
|
||||
FEXCore::IR::RegisterClassType Class {uint32_t(RegClass >> 32)};
|
||||
uint32_t Reg = RegClass;
|
||||
switch (Class) {
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
@@ -202,8 +207,11 @@ void Dump(std::stringstream *out, IRListView<false> const* IR, IR::RegisterAlloc
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
}
|
||||
|
||||
*out << std::dec << Reg << ")";
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
|
||||
} else {
|
||||
*out << ")";
|
||||
}
|
||||
}
|
||||
|
||||
*out << " i" << std::dec << (ElementSize * 8);
|
||||
@@ -245,9 +253,9 @@ void Dump(std::stringstream *out, IRListView<false> const* IR, IR::RegisterAlloc
|
||||
auto [NodeNode, IROp] = NodeBegin();
|
||||
auto PhiOp = IROp->C<IR::IROp_PhiValue>();
|
||||
*out << "[ ";
|
||||
PrintArg(out, IR, PhiOp->Value, RAPass);
|
||||
PrintArg(out, IR, PhiOp->Value, RAData);
|
||||
*out << ", ";
|
||||
PrintArg(out, IR, PhiOp->Block, RAPass);
|
||||
PrintArg(out, IR, PhiOp->Block, RAData);
|
||||
*out << " ]";
|
||||
|
||||
if (PhiOp->Next.ID())
|
||||
+604
@@ -0,0 +1,604 @@
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <istream>
|
||||
#include <unordered_map>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
|
||||
|
||||
|
||||
namespace FEXCore::IR {
|
||||
namespace {
|
||||
|
||||
enum class DecodeFailure {
|
||||
DECODE_OKAY,
|
||||
DECODE_UNKNOWN_TYPE,
|
||||
DECODE_INVALID,
|
||||
DECODE_INVALIDCHAR,
|
||||
DECODE_INVALIDRANGE,
|
||||
DECODE_INVALIDREGISTERCLASS,
|
||||
DECODE_UNKNOWN_SSA,
|
||||
DECODE_INVALID_CONDFLAG,
|
||||
DECODE_INVALID_MEMOFFSETTYPE,
|
||||
DECODE_INVALID_FENCETYPE,
|
||||
};
|
||||
|
||||
std::string ltrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(0, pos);
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
std::string rtrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
std::string trim(std::string String) {
|
||||
return rtrim(ltrim(String));
|
||||
}
|
||||
|
||||
std::string DecodeErrorToString(DecodeFailure Failure) {
|
||||
switch (Failure) {
|
||||
case DecodeFailure::DECODE_OKAY: return "Okay";
|
||||
case DecodeFailure::DECODE_UNKNOWN_TYPE: return "Unknown Type";
|
||||
case DecodeFailure::DECODE_INVALID: return "Invalid";
|
||||
case DecodeFailure::DECODE_INVALIDCHAR: return "Invalid starting char";
|
||||
case DecodeFailure::DECODE_INVALIDRANGE: return "Invalid integer range";
|
||||
case DecodeFailure::DECODE_INVALIDREGISTERCLASS: return "Invalid register class";
|
||||
case DecodeFailure::DECODE_UNKNOWN_SSA: return "Unknown SSA value";
|
||||
case DecodeFailure::DECODE_INVALID_CONDFLAG: return "Invalid Conditional name";
|
||||
case DecodeFailure::DECODE_INVALID_MEMOFFSETTYPE: return "Invalid Memory Offset Type";
|
||||
case DecodeFailure::DECODE_INVALID_FENCETYPE: return "Invalid Fence Type";
|
||||
};
|
||||
}
|
||||
|
||||
std::unordered_map<std::string_view, FEXCore::IR::IROps> NameToOpMap;
|
||||
|
||||
class IRParser: public FEXCore::IR::IREmitter {
|
||||
public:
|
||||
template<typename Type>
|
||||
std::pair<DecodeFailure, Type> DecodeValue(std::string &Arg) {
|
||||
return {DecodeFailure::DECODE_UNKNOWN_TYPE, {}};
|
||||
}
|
||||
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint8_t> DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint8_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint16_t> DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint16_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint32_t> DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint32_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint64_t> DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint64_t Result = strtoull(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::RegisterClassType> DecodeValue(std::string &Arg) {
|
||||
if (Arg == "GPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRClass};
|
||||
}
|
||||
else if (Arg == "FPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRClass};
|
||||
}
|
||||
else if (Arg == "GPRPair") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRPairClass};
|
||||
}
|
||||
else if (Arg == "Complex") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::ComplexClass};
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_INVALIDREGISTERCLASS, FEXCore::IR::InvalidClass};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::TypeDefinition> DecodeValue(std::string &Arg) {
|
||||
uint8_t Size{}, Elements{1};
|
||||
int NumArgs = sscanf(Arg.c_str(), "i%hhdv%hhd", &Size, &Elements);
|
||||
|
||||
if (NumArgs != 1 && NumArgs != 2) {
|
||||
return {DecodeFailure::DECODE_INVALID, {}};
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::TypeDefinition::Create(Size / 8, Elements)};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::CondClassType> DecodeValue(std::string &Arg) {
|
||||
std::array<std::string, 22> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
"UGE",
|
||||
"ULT",
|
||||
"MI",
|
||||
"PL",
|
||||
"VS",
|
||||
"VC",
|
||||
"UGT",
|
||||
"ULE",
|
||||
"SGE",
|
||||
"SLT",
|
||||
"SGT",
|
||||
"SLE",
|
||||
"Invalid Cond",
|
||||
"Invalid Cond",
|
||||
"FLU",
|
||||
"FGE",
|
||||
"FLEU",
|
||||
"FGT",
|
||||
"FU",
|
||||
"FNU"
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < CondNames.size(); ++i) {
|
||||
if (CondNames[i] == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, CondClassType{static_cast<uint8_t>(i)}};
|
||||
}
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_CONDFLAG, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::MemOffsetType> DecodeValue(std::string &Arg) {
|
||||
std::array<std::string, 3> Names = {
|
||||
"SXTX",
|
||||
"UXTW",
|
||||
"SXTW",
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < Names.size(); ++i) {
|
||||
if (Names[i] == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, MemOffsetType{static_cast<uint8_t>(i)}};
|
||||
}
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_MEMOFFSETTYPE, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::FenceType> DecodeValue(std::string &Arg) {
|
||||
std::array<std::string, 3> Names = {
|
||||
"Loads",
|
||||
"Stores",
|
||||
"LoadStores",
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < Names.size(); ++i) {
|
||||
if (Names[i] == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, FenceType{static_cast<uint8_t>(i)}};
|
||||
}
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_FENCETYPE, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, OrderedNode*> DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '%') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
// Strip off the type qualifier from the ssa value
|
||||
size_t ArgEnd = std::string::npos;
|
||||
std::string SSAName = trim(Arg);
|
||||
ArgEnd = SSAName.find_first_of(" ");
|
||||
|
||||
if (ArgEnd != std::string::npos) {
|
||||
SSAName = SSAName.substr(0, ArgEnd);
|
||||
}
|
||||
|
||||
// Forward declarations may make this not succed
|
||||
auto Op = SSANameMapper.find(SSAName);
|
||||
if (Op == SSANameMapper.end()) {
|
||||
return {DecodeFailure::DECODE_UNKNOWN_SSA, nullptr};
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_OKAY, Op->second};
|
||||
}
|
||||
|
||||
struct LineDefinition {
|
||||
size_t LineNumber;
|
||||
bool HasDefinition{};
|
||||
std::string Definition{};
|
||||
FEXCore::IR::TypeDefinition Size{};
|
||||
std::string IROp{};
|
||||
FEXCore::IR::IROps OpEnum;
|
||||
bool HasArgs{};
|
||||
std::vector<std::string> Args;
|
||||
OrderedNode *Node{};
|
||||
};
|
||||
|
||||
std::vector<std::string> Lines;
|
||||
std::unordered_map<std::string, OrderedNode*> SSANameMapper;
|
||||
std::vector<LineDefinition> Defs;
|
||||
LineDefinition *CurrentDef{};
|
||||
|
||||
IRParser(std::istream *text) {
|
||||
InitializeStaticTables();
|
||||
|
||||
std::string TmpLine;
|
||||
while (!text->eof()) {
|
||||
std::getline(*text, TmpLine);
|
||||
if (text->eof()) {
|
||||
break;
|
||||
}
|
||||
if (text->fail()) {
|
||||
LogMan::Msg::E("Failed to getline on line: %ld", Lines.size());
|
||||
return;
|
||||
}
|
||||
Lines.emplace_back(TmpLine);
|
||||
}
|
||||
|
||||
ResetWorkingList();
|
||||
Loaded = Parse();
|
||||
}
|
||||
|
||||
bool Loaded = false;
|
||||
|
||||
#define IROP_PARSER_ALLOCATE_HELPERS
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
|
||||
|
||||
bool Parse() {
|
||||
auto CheckPrintError = [&](LineDefinition &Def, DecodeFailure Failure) -> bool {
|
||||
if (Failure != DecodeFailure::DECODE_OKAY) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("Value Couldn't be decoded due to %s", DecodeErrorToString(Failure).c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
// String parse every line for our definitions
|
||||
for (size_t i = 0; i < Lines.size(); ++i) {
|
||||
std::string Line = Lines[i];
|
||||
LineDefinition Def{};
|
||||
CurrentDef = &Def;
|
||||
Def.LineNumber = i;
|
||||
|
||||
Line = trim(Line);
|
||||
|
||||
// Skip empty lines
|
||||
if (Line.empty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Line[0] == ';') {
|
||||
// This is a comment line
|
||||
// Skip it
|
||||
continue;
|
||||
}
|
||||
|
||||
size_t CurrentPos{};
|
||||
// Let's see if this node is assigning something first
|
||||
if (Line[0] == '%') {
|
||||
size_t DefinitionEnd = std::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of("=", CurrentPos)) != std::string::npos) {
|
||||
Def.Definition = Line.substr(0, DefinitionEnd);
|
||||
Def.Definition = trim(Def.Definition);
|
||||
Def.HasDefinition = true;
|
||||
CurrentPos = DefinitionEnd + 1; // +1 to ensure we go past then assignment
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("SSA declaration without assignment");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we are pulling in some IR from the IR Printer
|
||||
// Prints (%ssa%d) at the start of lines without a definition
|
||||
if (Line[0] == '(') {
|
||||
size_t DefinitionEnd = std::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of(")", CurrentPos)) != std::string::npos) {
|
||||
size_t SSAEnd = std::string::npos;
|
||||
if ((SSAEnd = Line.find_last_of(" ", DefinitionEnd)) != std::string::npos) {
|
||||
std::string Type = Line.substr(SSAEnd + 1, DefinitionEnd - SSAEnd - 1);
|
||||
Type = trim(Type);
|
||||
|
||||
auto DefinitionSize = DecodeValue<FEXCore::IR::TypeDefinition>(Type);
|
||||
if (!CheckPrintError(Def, DefinitionSize.first)) return false;
|
||||
Def.Size = DefinitionSize.second;
|
||||
}
|
||||
|
||||
Def.Definition = trim(Line.substr(1, std::min(DefinitionEnd, SSAEnd) - 1));
|
||||
|
||||
CurrentPos = DefinitionEnd + 1;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("SSA value with numbered SSA provided but no closing parentheses");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (Def.HasDefinition) {
|
||||
// Let's check if we have a size declared with this variable
|
||||
size_t NameEnd = std::string::npos;
|
||||
if ((NameEnd = Def.Definition.find_first_of(" ")) != std::string::npos) {
|
||||
std::string Type = Def.Definition.substr(NameEnd + 1);
|
||||
Type = trim(Type);
|
||||
Def.Definition = trim(Def.Definition.substr(0, NameEnd));
|
||||
|
||||
auto DefinitionSize = DecodeValue<FEXCore::IR::TypeDefinition>(Type);
|
||||
if (!CheckPrintError(Def, DefinitionSize.first)) return false;
|
||||
Def.Size = DefinitionSize.second;
|
||||
}
|
||||
|
||||
if (Def.Definition == "%Invalid") {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("Definition tried to define reserved %Invalid ssa node");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Let's get the IR op
|
||||
size_t OpNameEnd = std::string::npos;
|
||||
std::string RemainingLine = trim(Line.substr(CurrentPos));
|
||||
CurrentPos = 0;
|
||||
if ((OpNameEnd = RemainingLine.find_first_of(" \t\n\r\0", CurrentPos)) != std::string::npos) {
|
||||
Def.IROp = RemainingLine.substr(CurrentPos, OpNameEnd);
|
||||
Def.IROp = trim(Def.IROp);
|
||||
Def.HasArgs = true;
|
||||
CurrentPos = OpNameEnd;
|
||||
}
|
||||
else {
|
||||
if (RemainingLine.empty()) {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("Line without an IROp?");
|
||||
return false;
|
||||
}
|
||||
|
||||
Def.IROp = RemainingLine;
|
||||
Def.HasArgs = false;
|
||||
}
|
||||
|
||||
if (Def.HasArgs) {
|
||||
RemainingLine = trim(RemainingLine.substr(CurrentPos));
|
||||
CurrentPos = 0;
|
||||
if (RemainingLine.empty()) {
|
||||
// How did we get here?
|
||||
Def.HasArgs = false;
|
||||
}
|
||||
else {
|
||||
while (!RemainingLine.empty()) {
|
||||
size_t ArgEnd = std::string::npos;
|
||||
ArgEnd = RemainingLine.find_first_of(",");
|
||||
|
||||
std::string Arg = RemainingLine.substr(0, ArgEnd);
|
||||
Arg = trim(Arg);
|
||||
Def.Args.emplace_back(Arg);
|
||||
|
||||
RemainingLine.erase(0, ArgEnd+1); // +1 to ensure we go past the ','
|
||||
if (ArgEnd == std::string::npos)
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Defs.emplace_back(Def);
|
||||
}
|
||||
|
||||
// Ensure all of the ops are real ops
|
||||
for(size_t i = 0; i < Defs.size(); ++i) {
|
||||
auto &Def = Defs[i];
|
||||
auto Op = NameToOpMap.find(Def.IROp);
|
||||
if (Op == NameToOpMap.end()) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("IROp '%s' doesn't exist", Def.IROp.c_str());
|
||||
return false;
|
||||
}
|
||||
Def.OpEnum = Op->second;
|
||||
}
|
||||
|
||||
// Emit the header op
|
||||
IRPair<IROp_IRHeader> IRHeader;
|
||||
{
|
||||
auto &Def = Defs[0];
|
||||
CurrentDef = &Def;
|
||||
if (Def.OpEnum != FEXCore::IR::IROps::OP_IRHEADER) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("First op needs to be IRHeader. Was '%s'", Def.IROp.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Entry = DecodeValue<uint64_t>(Def.Args[0]);
|
||||
auto CodeBlockCount = DecodeValue<uint64_t>(Def.Args[2]);
|
||||
|
||||
if (!CheckPrintError(Def, Entry.first)) return false;
|
||||
if (!CheckPrintError(Def, CodeBlockCount.first)) return false;
|
||||
|
||||
IRHeader = _IRHeader(InvalidNode, Entry.second, CodeBlockCount.second, false);
|
||||
}
|
||||
|
||||
SetWriteCursor(nullptr); // isolate the header from everything following
|
||||
|
||||
// Initialize SSANameMapper with Invalid value
|
||||
SSANameMapper["%Invalid"] = Invalid();
|
||||
|
||||
// Spin through the blocks and generate basic block ops
|
||||
for(size_t i = 0; i < Defs.size(); ++i) {
|
||||
auto &Def = Defs[i];
|
||||
if (Def.OpEnum == FEXCore::IR::IROps::OP_CODEBLOCK) {
|
||||
auto CodeBlock = _CodeBlock(InvalidNode, InvalidNode);
|
||||
SSANameMapper[Def.Definition] = CodeBlock.Node;
|
||||
Def.Node = CodeBlock.Node;
|
||||
|
||||
if (i == 1) {
|
||||
// First code block is the entry block
|
||||
// Link the header to the first block
|
||||
IRHeader.first->Blocks = CodeBlock.Node->Wrapped(ListData.Begin());
|
||||
}
|
||||
CodeBlocks.emplace_back(CodeBlock.Node);
|
||||
}
|
||||
}
|
||||
SetWriteCursor(nullptr); // isolate the block headers too
|
||||
|
||||
// Spin through all the definitions and add the ops to the basic blocks
|
||||
OrderedNode *CurrentBlock{};
|
||||
FEXCore::IR::IROp_CodeBlock *CurrentBlockOp{};
|
||||
for(size_t i = 1; i < Defs.size(); ++i) {
|
||||
auto &Def = Defs[i];
|
||||
CurrentDef = &Def;
|
||||
|
||||
|
||||
switch (Def.OpEnum) {
|
||||
// Special handled
|
||||
case FEXCore::IR::IROps::OP_IRHEADER:
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("IRHEADER used in the middle of the block!");
|
||||
return false; // only one OP_IRHEADER allowed per block
|
||||
|
||||
case FEXCore::IR::IROps::OP_CODEBLOCK: {
|
||||
SetWriteCursor(nullptr); // isolate from previous block
|
||||
if (CurrentBlock != nullptr) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("CodeBlock being used inside of already existing codeblock!");
|
||||
return false;
|
||||
}
|
||||
|
||||
CurrentBlock = Def.Node;
|
||||
CurrentBlockOp = CurrentBlock->Op(Data.Begin())->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
case FEXCore::IR::IROps::OP_BEGINBLOCK: {
|
||||
if (CurrentBlock == nullptr) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("EndBlock being used outside of a block!");
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Adjust = DecodeValue<OrderedNode*>(Def.Args[0]);
|
||||
|
||||
if (!CheckPrintError(Def, Adjust.first)) return false;
|
||||
|
||||
Def.Node = _BeginBlock(Adjust.second);
|
||||
CurrentBlockOp->Begin = Def.Node->Wrapped(ListData.Begin());
|
||||
break;
|
||||
}
|
||||
|
||||
case FEXCore::IR::IROps::OP_ENDBLOCK: {
|
||||
if (CurrentBlock == nullptr) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("EndBlock being used outside of a block!");
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Adjust = DecodeValue<OrderedNode*>(Def.Args[0]);
|
||||
|
||||
if (!CheckPrintError(Def, Adjust.first)) return false;
|
||||
|
||||
Def.Node = _EndBlock(Adjust.second);
|
||||
CurrentBlockOp->Last = Def.Node->Wrapped(ListData.Begin());
|
||||
|
||||
CurrentBlock = nullptr;
|
||||
CurrentBlockOp = nullptr;
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case FEXCore::IR::IROps::OP_DUMMY: {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("Dummy op must not be used");
|
||||
|
||||
break;
|
||||
}
|
||||
#define IROP_PARSER_SWITCH_HELPERS
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
default: {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("Unhandled Op enum '%s' in parser", Def.IROp.c_str());
|
||||
return false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Def.HasDefinition) {
|
||||
auto IROp = Def.Node->Op(Data.Begin());
|
||||
if (Def.Size.Elements()) {
|
||||
IROp->Size = Def.Size.Bytes() * Def.Size.Elements();
|
||||
IROp->ElementSize = Def.Size.Bytes();
|
||||
}
|
||||
else {
|
||||
IROp->Size = Def.Size.Bytes();
|
||||
IROp->ElementSize = 0;
|
||||
}
|
||||
SSANameMapper[Def.Definition] = Def.Node;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void InitializeStaticTables() {
|
||||
if (NameToOpMap.size() == 0) {
|
||||
for (FEXCore::IR::IROps Op = FEXCore::IR::IROps::OP_DUMMY;
|
||||
Op <= FEXCore::IR::IROps::OP_LAST;
|
||||
Op = static_cast<FEXCore::IR::IROps>(static_cast<uint32_t>(Op) + 1)) {
|
||||
NameToOpMap[FEXCore::IR::GetName(Op)] = Op;
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
} // anon namespace
|
||||
|
||||
IREmitter* Parse(std::istream *in) {
|
||||
auto parser = new IRParser(in);
|
||||
|
||||
if (parser->Loaded) {
|
||||
return parser;
|
||||
} else {
|
||||
delete parser;
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
+2
-4
@@ -1,6 +1,6 @@
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
@@ -11,9 +11,7 @@ void PassManager::AddDefaultPasses(bool InlineConstants, bool StaticRegisterAllo
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateContextLoadStoreElimination());
|
||||
InsertPass(CreateDeadFlagStoreElimination());
|
||||
InsertPass(CreateDeadGPRStoreElimination());
|
||||
InsertPass(CreateDeadFPRStoreElimination());
|
||||
InsertPass(CreateDeadStoreElimination());
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
InsertPass(CreateConstProp(InlineConstants));
|
||||
|
||||
|
||||
+2
-3
@@ -3,14 +3,13 @@
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
FEXCore::IR::Pass* CreateConstProp(bool InlineConstants);
|
||||
FEXCore::IR::Pass* CreateContextLoadStoreElimination();
|
||||
FEXCore::IR::Pass* CreateSyscallOptimization();
|
||||
FEXCore::IR::Pass* CreateDeadFlagCalculationEliminination();
|
||||
FEXCore::IR::Pass* CreateDeadFlagStoreElimination();
|
||||
FEXCore::IR::Pass* CreateDeadGPRStoreElimination();
|
||||
FEXCore::IR::Pass* CreateDeadFPRStoreElimination();
|
||||
FEXCore::IR::Pass* CreateDeadStoreElimination();
|
||||
FEXCore::IR::Pass* CreatePassDeadCodeElimination();
|
||||
FEXCore::IR::Pass* CreateIRCompaction();
|
||||
FEXCore::IR::RegisterAllocationPass* CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool OptimizeSRA);
|
||||
|
||||
@@ -12,6 +12,8 @@
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class ConstProp final : public FEXCore::IR::Pass {
|
||||
std::unordered_map<uint64_t, OrderedNode*> ConstPool;
|
||||
std::map<OrderedNode*, uint64_t> AddressgenConsts;
|
||||
public:
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
bool InlineConstants;
|
||||
@@ -164,22 +166,21 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
auto HeaderOp = CurrentIR.GetHeader();
|
||||
|
||||
{
|
||||
std::map<uint64_t, OrderedNode*> Consts;
|
||||
|
||||
// constants are pooled per block
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
if (Consts.count(Op->Constant)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Consts[Op->Constant]);
|
||||
if (ConstPool.count(Op->Constant)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ConstPool[Op->Constant]);
|
||||
Changed = true;
|
||||
} else {
|
||||
Consts[Op->Constant] = CodeNode;
|
||||
ConstPool[Op->Constant] = CodeNode;
|
||||
}
|
||||
}
|
||||
}
|
||||
Consts.clear();
|
||||
ConstPool.clear();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -265,14 +266,13 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
|
||||
// LoadMem / StoreMem imm pooling
|
||||
// If imms are close by, use address gen to generate the values instead of using a new imm
|
||||
std::map<OrderedNode*, uint64_t> Consts;
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_LOADMEM || IROp->Op == OP_STOREMEM) {
|
||||
uint64_t Addr;
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Addr) && IROp->Args[1].IsInvalid()) {
|
||||
for (auto& Const: Consts) {
|
||||
for (auto& Const: AddressgenConsts) {
|
||||
if ((Addr - Const.second) < 65536) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, Const.first);
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_Constant(Addr - Const.second));
|
||||
@@ -280,14 +280,14 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
Consts[IREmit->UnwrapNode(IROp->Args[0])] = Addr;
|
||||
AddressgenConsts[IREmit->UnwrapNode(IROp->Args[0])] = Addr;
|
||||
}
|
||||
doneOp:
|
||||
;
|
||||
}
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
}
|
||||
Consts.clear();
|
||||
AddressgenConsts.clear();
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
|
||||
+31
-27
@@ -28,7 +28,10 @@ namespace {
|
||||
FEXCore::IR::OrderedNode *StoreNode;
|
||||
};
|
||||
|
||||
using ContextInfo = std::vector<ContextMemberInfo>;
|
||||
struct ContextInfo {
|
||||
std::vector<ContextMemberInfo*> Lookup;
|
||||
std::vector<ContextMemberInfo> ClassificationInfo;
|
||||
};
|
||||
|
||||
constexpr static std::array<LastAccessType, 14> DefaultAccess = {
|
||||
ACCESS_NONE,
|
||||
@@ -47,7 +50,9 @@ namespace {
|
||||
ACCESS_NONE,
|
||||
};
|
||||
|
||||
static void ClassifyContextStruct(ContextInfo *ContextClassification) {
|
||||
static void ClassifyContextStruct(ContextInfo *ContextClassificationInfo) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
@@ -88,15 +93,6 @@ namespace {
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
sizeof(FEXCore::Core::CPUState::gs),
|
||||
},
|
||||
DefaultAccess[4],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es),
|
||||
@@ -133,6 +129,15 @@ namespace {
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
sizeof(FEXCore::Core::CPUState::gs),
|
||||
},
|
||||
DefaultAccess[4],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
@@ -186,16 +191,27 @@ namespace {
|
||||
}
|
||||
|
||||
size_t ClassifiedStructSize{};
|
||||
ContextClassificationInfo->Lookup.reserve(sizeof(FEXCore::Core::CPUState));
|
||||
for (auto &it : *ContextClassification) {
|
||||
LogMan::Throw::A(it.Class.Offset == ContextClassificationInfo->Lookup.size(), "Offset missmatch %d %d", it.Class.Offset == ContextClassificationInfo->Lookup.size());
|
||||
for (int i = 0; i < it.Class.Size; i++) {
|
||||
ContextClassificationInfo->Lookup.push_back(&it);
|
||||
}
|
||||
ClassifiedStructSize += it.Class.Size;
|
||||
}
|
||||
|
||||
LogMan::Throw::A(ClassifiedStructSize == sizeof(FEXCore::Core::CPUState),
|
||||
"Classified CPUStruct size doesn't match real CPUState struct size! %ld != %ld",
|
||||
ClassifiedStructSize, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
LogMan::Throw::A(ContextClassificationInfo->Lookup.size() == sizeof(FEXCore::Core::CPUState),
|
||||
"Classified CPUStruct size doesn't match real CPUState struct size! %ld != %ld",
|
||||
ContextClassificationInfo->Lookup.size(), sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
static void ResetClassificationAccesses(ContextInfo *ContextClassification) {
|
||||
static void ResetClassificationAccesses(ContextInfo *ContextClassificationInfo) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
auto SetAccess = [&](size_t Offset, auto Access) {
|
||||
ContextClassification->at(Offset).Accessed = Access;
|
||||
ContextClassification->at(Offset).AccessRegClass = FEXCore::IR::InvalidClass;
|
||||
@@ -265,20 +281,8 @@ private:
|
||||
bool RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit);
|
||||
};
|
||||
|
||||
ContextMemberInfo *RCLSE::FindMemberInfo(ContextInfo *ClassifiedInfo, uint32_t Offset, uint8_t Size) {
|
||||
ContextMemberInfo *Info{};
|
||||
// Just linearly scan to find the info
|
||||
for (size_t i = 0; i < ClassifiedInfo->size(); ++i) {
|
||||
ContextMemberInfo *LocalInfo = &ClassifiedInfo->at(i);
|
||||
if (LocalInfo->Class.Offset <= Offset &&
|
||||
(LocalInfo->Class.Offset + LocalInfo->Class.Size) > Offset) {
|
||||
Info = LocalInfo;
|
||||
break;
|
||||
}
|
||||
}
|
||||
LogMan::Throw::A(Info != nullptr, "Couldn't find Context Member to record to");
|
||||
|
||||
return Info;
|
||||
ContextMemberInfo *RCLSE::FindMemberInfo(ContextInfo *ContextClassificationInfo, uint32_t Offset, uint8_t Size) {
|
||||
return ContextClassificationInfo->Lookup.at(Offset);
|
||||
}
|
||||
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
@@ -404,7 +408,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
|
||||
// XXX: Walk the list and calculate the control flow
|
||||
|
||||
ContextInfo LocalInfo = ClassifiedStruct;
|
||||
ContextInfo &LocalInfo = ClassifiedStruct;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
|
||||
|
||||
@@ -1,170 +0,0 @@
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
// Higher values might result in more stores getting eliminated but will make the optimization take more time
|
||||
constexpr int PropagationRounds = 5;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class DeadFPRStoreElimination final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
};
|
||||
|
||||
struct FPRInfo {
|
||||
uint64_t reads { 0 };
|
||||
uint64_t writes { 0 };
|
||||
uint64_t kill { 0 };
|
||||
};
|
||||
|
||||
bool IsFPR(uint32_t Offset) {
|
||||
|
||||
auto begin = offsetof(FEXCore::Core::ThreadState, State.xmm[0][0]);
|
||||
auto end = offsetof(FEXCore::Core::ThreadState, State.xmm[17][0]);
|
||||
|
||||
if (Offset < begin || Offset >= end)
|
||||
return false;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool IsTrackedWriteFPR(uint32_t Offset, uint8_t Size) {
|
||||
if (Size != 16 && Size != 8 && Size != 4)
|
||||
return false;
|
||||
if (Offset & 15)
|
||||
return false;
|
||||
|
||||
return IsFPR(Offset);
|
||||
}
|
||||
|
||||
|
||||
|
||||
uint64_t FPRBit(uint32_t Offset, uint32_t Size) {
|
||||
if (!IsFPR(Offset)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
auto begin = offsetof(FEXCore::Core::ThreadState, State.xmm[0][0]);
|
||||
|
||||
auto regn = (Offset - begin)/16;
|
||||
auto bitn = regn * 3;
|
||||
|
||||
if (!IsTrackedWriteFPR(Offset, Size))
|
||||
return 7UL << (bitn);
|
||||
|
||||
if (Size == 16)
|
||||
return 7UL << (bitn);
|
||||
else if (Size == 8)
|
||||
return 3UL << (bitn);
|
||||
else if (Size == 4)
|
||||
return 1UL << (bitn);
|
||||
else
|
||||
LogMan::Throw::A(false, "Unexpected FPR size %d", Size);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief This is a temporary pass to detect simple multiblock dead FPR stores
|
||||
*
|
||||
* First pass computes which FPRs are read and written per block
|
||||
*
|
||||
* Second pass computes which FPRs are stored, but overwritten by the next block(s).
|
||||
* It also propagates this information a few times to catch dead FPRs across multiple blocks.
|
||||
*
|
||||
* Third pass removes the dead stores.
|
||||
*
|
||||
*/
|
||||
bool DeadFPRStoreElimination::Run(IREmitter *IREmit) {
|
||||
std::map<OrderedNode*, FPRInfo> FPRMap;
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
// Pass 1
|
||||
// Compute FPRs read/writes per block
|
||||
// This is conservative and doesn't try to be smart about loads after writes
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreContext>();
|
||||
if (IsTrackedWriteFPR(Op->Offset, IROp->Size))
|
||||
FPRMap[BlockNode].writes |= FPRBit(Op->Offset, IROp->Size);
|
||||
else
|
||||
FPRMap[BlockNode].reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
}
|
||||
else if (IROp->Op == OP_STORECONTEXTINDEXED ||
|
||||
IROp->Op == OP_LOADCONTEXTINDEXED) {
|
||||
// We can't track through these
|
||||
FPRMap[BlockNode].reads = -1;
|
||||
}
|
||||
else if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadContext>();
|
||||
FPRMap[BlockNode].reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2
|
||||
// Compute FPRs that are stored, but always ovewritten in the next blocks
|
||||
// Propagate the information a few times to eliminate more
|
||||
for (int i = 0; i < PropagationRounds; i++)
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
if (IROp->Op == OP_JUMP) {
|
||||
auto Op = IROp->CW<IR::IROp_Jump>();
|
||||
OrderedNode *TargetNode = CurrentIR.GetNode(Op->Header.Args[0]);
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
FPRMap[BlockNode].kill = FPRMap[TargetNode].writes & ~(FPRMap[TargetNode].reads) & ~FPRMap[BlockNode].reads;
|
||||
|
||||
// FPRs that are written by the next block can be considered as written by this block, if not read
|
||||
FPRMap[BlockNode].writes |= FPRMap[BlockNode].kill & ~FPRMap[BlockNode].reads;
|
||||
}
|
||||
else if (IROp->Op == OP_CONDJUMP) {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
OrderedNode *TrueTargetNode = CurrentIR.GetNode(Op->TrueBlock);
|
||||
OrderedNode *FalseTargetNode = CurrentIR.GetNode(Op->FalseBlock);
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
FPRMap[BlockNode].kill = FPRMap[TrueTargetNode].writes & ~(FPRMap[TrueTargetNode].reads) & ~FPRMap[BlockNode].reads;
|
||||
FPRMap[BlockNode].kill &= FPRMap[FalseTargetNode].writes & ~(FPRMap[FalseTargetNode].reads) & ~FPRMap[BlockNode].reads;
|
||||
|
||||
// FPRs that are written by the next blocks can be considered as written by this block, if not read
|
||||
FPRMap[BlockNode].writes |= FPRMap[BlockNode].kill & ~FPRMap[BlockNode].reads;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 3
|
||||
// Remove the dead stores
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreContext>();
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if ((FPRMap[BlockNode].kill & FPRBit(Op->Offset, IROp->Size)) == FPRBit(Op->Offset, IROp->Size) && (FPRBit(Op->Offset, IROp->Size) != 0)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateDeadFPRStoreElimination() {
|
||||
return new DeadFPRStoreElimination{};
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,119 +0,0 @@
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class DeadFlagStoreElimination final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
};
|
||||
|
||||
struct FlagInfo {
|
||||
uint64_t reads { 0 };
|
||||
uint64_t writes { 0 };
|
||||
uint64_t kill { 0 };
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is a temporary pass to detect simple multiblock dead flag stores
|
||||
*
|
||||
* First pass computes which flags are read and written per block
|
||||
*
|
||||
* Second pass computes which flags are stored, but overwritten by the next block(s).
|
||||
* It also propagates this information a few times to catch dead flags across multiple blocks.
|
||||
*
|
||||
* Third pass removes the dead stores.
|
||||
*
|
||||
*/
|
||||
bool DeadFlagStoreElimination::Run(IREmitter *IREmit) {
|
||||
std::map<OrderedNode*, FlagInfo> FlagMap;
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
// Pass 1
|
||||
// Compute flags read/writes per block
|
||||
// This is conservative and doesn't try to be smart about loads after writes
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreFlag>();
|
||||
FlagMap[BlockNode].writes |= 1UL << Op->Flag;
|
||||
}
|
||||
else if (IROp->Op == OP_INVALIDATEFLAGS) {
|
||||
auto Op = IROp->CW<IR::IROp_InvalidateFlags>();
|
||||
FlagMap[BlockNode].writes |= Op->Flags;
|
||||
}
|
||||
else if (IROp->Op == OP_LOADFLAG) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadFlag>();
|
||||
FlagMap[BlockNode].reads |= 1UL << Op->Flag;
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2
|
||||
// Compute flags that are stored, but always ovewritten in the next blocks
|
||||
// Propagate the information a few times to eliminate more
|
||||
for (int i = 0; i < 5; i++)
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
if (IROp->Op == OP_JUMP) {
|
||||
auto Op = IROp->CW<IR::IROp_Jump>();
|
||||
OrderedNode *TargetNode = CurrentIR.GetNode(Op->Header.Args[0]);
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
FlagMap[BlockNode].kill = FlagMap[TargetNode].writes & ~(FlagMap[TargetNode].reads) & ~FlagMap[BlockNode].reads;
|
||||
|
||||
// Flags that are written by the next block can be considered as written by this block, if not read
|
||||
FlagMap[BlockNode].writes |= FlagMap[BlockNode].kill & ~FlagMap[BlockNode].reads;
|
||||
}
|
||||
else if (IROp->Op == OP_CONDJUMP) {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
OrderedNode *TrueTargetNode = CurrentIR.GetNode(Op->TrueBlock);
|
||||
OrderedNode *FalseTargetNode = CurrentIR.GetNode(Op->FalseBlock);
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
FlagMap[BlockNode].kill = FlagMap[TrueTargetNode].writes & ~(FlagMap[TrueTargetNode].reads) & ~FlagMap[BlockNode].reads;
|
||||
FlagMap[BlockNode].kill &= FlagMap[FalseTargetNode].writes & ~(FlagMap[FalseTargetNode].reads) & ~FlagMap[BlockNode].reads;
|
||||
|
||||
// Flags that are written by the next blocks can be considered as written by this block, if not read
|
||||
FlagMap[BlockNode].writes |= FlagMap[BlockNode].kill & ~FlagMap[BlockNode].reads;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 3
|
||||
// Remove the dead stores
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreFlag>();
|
||||
// If this StoreFlag is never read, remove it
|
||||
if (FlagMap[BlockNode].kill & (1UL << Op->Flag)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateDeadFlagStoreElimination() {
|
||||
return new DeadFlagStoreElimination{};
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,154 +0,0 @@
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
// Higher values might result in more stores getting eliminated but will make the optimization take more time
|
||||
constexpr int PropagationRounds = 5;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class DeadGPRStoreElimination final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
};
|
||||
|
||||
struct GPRInfo {
|
||||
uint32_t reads { 0 };
|
||||
uint32_t writes { 0 };
|
||||
uint32_t kill { 0 };
|
||||
};
|
||||
|
||||
bool IsFullGPR(uint32_t Offset, uint8_t Size) {
|
||||
if (Size != 8)
|
||||
return false;
|
||||
if (Offset & 7)
|
||||
return false;
|
||||
|
||||
if (Offset < 8 || Offset >= (17 * 8))
|
||||
return false;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool IsGPR(uint32_t Offset) {
|
||||
|
||||
if (Offset < 8 || Offset >= (17 * 8))
|
||||
return false;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
uint32_t GPRBit(uint32_t Offset) {
|
||||
if (!IsGPR(Offset)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
return 1 << ((Offset - 8)/8);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief This is a temporary pass to detect simple multiblock dead GPR stores
|
||||
*
|
||||
* First pass computes which GPRs are read and written per block
|
||||
*
|
||||
* Second pass computes which GPRs are stored, but overwritten by the next block(s).
|
||||
* It also propagates this information a few times to catch dead GPRs across multiple blocks.
|
||||
*
|
||||
* Third pass removes the dead stores.
|
||||
*
|
||||
*/
|
||||
bool DeadGPRStoreElimination::Run(IREmitter *IREmit) {
|
||||
std::map<OrderedNode*, GPRInfo> GPRMap;
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
// Pass 1
|
||||
// Compute GPRs read/writes per block
|
||||
// This is conservative and doesn't try to be smart about loads after writes
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreContext>();
|
||||
if (IsFullGPR(Op->Offset, IROp->Size))
|
||||
GPRMap[BlockNode].writes |= GPRBit(Op->Offset);
|
||||
else
|
||||
GPRMap[BlockNode].reads |= GPRBit(Op->Offset);
|
||||
}
|
||||
else if (IROp->Op == OP_STORECONTEXTINDEXED ||
|
||||
IROp->Op == OP_LOADCONTEXTINDEXED) {
|
||||
// We can't track through these
|
||||
GPRMap[BlockNode].reads = -1;
|
||||
}
|
||||
else if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadContext>();
|
||||
GPRMap[BlockNode].reads |= GPRBit(Op->Offset);
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2
|
||||
// Compute GPRs that are stored, but always ovewritten in the next blocks
|
||||
// Propagate the information a few times to eliminate more
|
||||
for (int i = 0; i < PropagationRounds; i++)
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
if (IROp->Op == OP_JUMP) {
|
||||
auto Op = IROp->CW<IR::IROp_Jump>();
|
||||
OrderedNode *TargetNode = CurrentIR.GetNode(Op->Header.Args[0]);
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
GPRMap[BlockNode].kill = GPRMap[TargetNode].writes & ~(GPRMap[TargetNode].reads) & ~GPRMap[BlockNode].reads;
|
||||
|
||||
// GPRs that are written by the next block can be considered as written by this block, if not read
|
||||
GPRMap[BlockNode].writes |= GPRMap[BlockNode].kill & ~GPRMap[BlockNode].reads;
|
||||
}
|
||||
else if (IROp->Op == OP_CONDJUMP) {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
OrderedNode *TrueTargetNode = CurrentIR.GetNode(Op->TrueBlock);
|
||||
OrderedNode *FalseTargetNode = CurrentIR.GetNode(Op->FalseBlock);
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
GPRMap[BlockNode].kill = GPRMap[TrueTargetNode].writes & ~(GPRMap[TrueTargetNode].reads) & ~GPRMap[BlockNode].reads;
|
||||
GPRMap[BlockNode].kill &= GPRMap[FalseTargetNode].writes & ~(GPRMap[FalseTargetNode].reads) & ~GPRMap[BlockNode].reads;
|
||||
|
||||
// GPRs that are written by the next blocks can be considered as written by this block, if not read
|
||||
GPRMap[BlockNode].writes |= GPRMap[BlockNode].kill & ~GPRMap[BlockNode].reads;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 3
|
||||
// Remove the dead stores
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreContext>();
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if (GPRMap[BlockNode].kill & GPRBit(Op->Offset)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateDeadGPRStoreElimination() {
|
||||
return new DeadGPRStoreElimination{};
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,331 @@
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
constexpr int PropagationRounds = 5;
|
||||
|
||||
class DeadStoreElimination final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
};
|
||||
|
||||
struct FlagInfo {
|
||||
uint64_t reads { 0 };
|
||||
uint64_t writes { 0 };
|
||||
uint64_t kill { 0 };
|
||||
};
|
||||
|
||||
|
||||
struct GPRInfo {
|
||||
uint32_t reads { 0 };
|
||||
uint32_t writes { 0 };
|
||||
uint32_t kill { 0 };
|
||||
};
|
||||
|
||||
bool IsFullGPR(uint32_t Offset, uint8_t Size) {
|
||||
if (Size != 8)
|
||||
return false;
|
||||
if (Offset & 7)
|
||||
return false;
|
||||
|
||||
if (Offset < 8 || Offset >= (17 * 8))
|
||||
return false;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool IsGPR(uint32_t Offset) {
|
||||
|
||||
if (Offset < 8 || Offset >= (17 * 8))
|
||||
return false;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
uint32_t GPRBit(uint32_t Offset) {
|
||||
if (!IsGPR(Offset)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
return 1 << ((Offset - 8)/8);
|
||||
}
|
||||
|
||||
struct FPRInfo {
|
||||
uint64_t reads { 0 };
|
||||
uint64_t writes { 0 };
|
||||
uint64_t kill { 0 };
|
||||
};
|
||||
|
||||
bool IsFPR(uint32_t Offset) {
|
||||
|
||||
auto begin = offsetof(FEXCore::Core::ThreadState, State.xmm[0][0]);
|
||||
auto end = offsetof(FEXCore::Core::ThreadState, State.xmm[17][0]);
|
||||
|
||||
if (Offset < begin || Offset >= end)
|
||||
return false;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool IsTrackedWriteFPR(uint32_t Offset, uint8_t Size) {
|
||||
if (Size != 16 && Size != 8 && Size != 4)
|
||||
return false;
|
||||
if (Offset & 15)
|
||||
return false;
|
||||
|
||||
return IsFPR(Offset);
|
||||
}
|
||||
|
||||
|
||||
|
||||
uint64_t FPRBit(uint32_t Offset, uint32_t Size) {
|
||||
if (!IsFPR(Offset)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
auto begin = offsetof(FEXCore::Core::ThreadState, State.xmm[0][0]);
|
||||
|
||||
auto regn = (Offset - begin)/16;
|
||||
auto bitn = regn * 3;
|
||||
|
||||
if (!IsTrackedWriteFPR(Offset, Size))
|
||||
return 7UL << (bitn);
|
||||
|
||||
if (Size == 16)
|
||||
return 7UL << (bitn);
|
||||
else if (Size == 8)
|
||||
return 3UL << (bitn);
|
||||
else if (Size == 4)
|
||||
return 1UL << (bitn);
|
||||
else
|
||||
LogMan::Msg::A("Unexpected FPR size %d", Size);
|
||||
|
||||
return 7UL << (bitn); // Return maximum on failure case
|
||||
}
|
||||
|
||||
struct Info {
|
||||
FlagInfo flag;
|
||||
GPRInfo gpr;
|
||||
FPRInfo fpr;
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* @brief This is a temporary pass to detect simple multiblock dead flag/gpr/fpr stores
|
||||
*
|
||||
* First pass computes which flags/gprs/fprs are read and written per block
|
||||
*
|
||||
* Second pass computes which flags/gprs/fprs are stored, but overwritten by the next block(s).
|
||||
* It also propagates this information a few times to catch dead flags/gprs/fprs across multiple blocks.
|
||||
*
|
||||
* Third pass removes the dead stores.
|
||||
*
|
||||
*/
|
||||
bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
std::unordered_map<OrderedNode*, Info> InfoMap;
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
// Pass 1
|
||||
// Compute flags/gprs/fprs read/writes per block
|
||||
// This is conservative and doesn't try to be smart about loads after writes
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.writes |= 1UL << Op->Flag;
|
||||
} else if (IROp->Op == OP_INVALIDATEFLAGS) {
|
||||
auto Op = IROp->C<IR::IROp_InvalidateFlags>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.writes |= Op->Flags;
|
||||
} else if (IROp->Op == OP_LOADFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.reads |= 1UL << Op->Flag;
|
||||
} else if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
if (IsFullGPR(Op->Offset, IROp->Size))
|
||||
BlockInfo.gpr.writes |= GPRBit(Op->Offset);
|
||||
else
|
||||
BlockInfo.gpr.reads |= GPRBit(Op->Offset);
|
||||
|
||||
//// FPR ////
|
||||
if (IsTrackedWriteFPR(Op->Offset, IROp->Size))
|
||||
BlockInfo.fpr.writes |= FPRBit(Op->Offset, IROp->Size);
|
||||
else
|
||||
BlockInfo.fpr.reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
} else if (IROp->Op == OP_STORECONTEXTINDEXED ||
|
||||
IROp->Op == OP_LOADCONTEXTINDEXED) {
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
// We can't track through these
|
||||
BlockInfo.gpr.reads = -1;
|
||||
|
||||
//// FPR ////
|
||||
// We can't track through these
|
||||
BlockInfo.fpr.reads = -1;
|
||||
} else if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
BlockInfo.gpr.reads |= GPRBit(Op->Offset);
|
||||
|
||||
//// FPR ////
|
||||
BlockInfo.fpr.reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2
|
||||
// Compute flags/gprs/fprs that are stored, but always ovewritten in the next blocks
|
||||
// Propagate the information a few times to eliminate more
|
||||
for (int i = 0; i < PropagationRounds; i++)
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto CodeBlock = BlockIROp->C<IROp_CodeBlock>();
|
||||
|
||||
auto IROp = CurrentIR.GetNode(CurrentIR.GetNode(CodeBlock->Last)->Header.Previous)->Op(CurrentIR.GetData());
|
||||
|
||||
if (IROp->Op == OP_JUMP) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
OrderedNode *TargetNode = CurrentIR.GetNode(Op->Header.Args[0]);
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
auto& TargetInfo = InfoMap[TargetNode];
|
||||
|
||||
//// Flags ////
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
BlockInfo.flag.kill = TargetInfo.flag.writes & ~(TargetInfo.flag.reads) & ~BlockInfo.flag.reads;
|
||||
|
||||
// Flags that are written by the next block can be considered as written by this block, if not read
|
||||
BlockInfo.flag.writes |= BlockInfo.flag.kill & ~BlockInfo.flag.reads;
|
||||
|
||||
|
||||
//// GPRs ////
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
BlockInfo.gpr.kill = TargetInfo.gpr.writes & ~(TargetInfo.gpr.reads) & ~BlockInfo.gpr.reads;
|
||||
|
||||
// GPRs that are written by the next block can be considered as written by this block, if not read
|
||||
BlockInfo.gpr.writes |= BlockInfo.gpr.kill & ~BlockInfo.gpr.reads;
|
||||
|
||||
|
||||
//// FPRs ////
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
BlockInfo.fpr.kill = TargetInfo.fpr.writes & ~(TargetInfo.fpr.reads) & ~BlockInfo.fpr.reads;
|
||||
|
||||
// FPRs that are written by the next block can be considered as written by this block, if not read
|
||||
BlockInfo.fpr.writes |= BlockInfo.fpr.kill & ~BlockInfo.fpr.reads;
|
||||
|
||||
} else if (IROp->Op == OP_CONDJUMP) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
OrderedNode *TrueTargetNode = CurrentIR.GetNode(Op->TrueBlock);
|
||||
OrderedNode *FalseTargetNode = CurrentIR.GetNode(Op->FalseBlock);
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
auto& TrueTargetInfo = InfoMap[TrueTargetNode];
|
||||
auto& FalseTargetInfo = InfoMap[FalseTargetNode];
|
||||
|
||||
//// Flags ////
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
BlockInfo.flag.kill = TrueTargetInfo.flag.writes & ~(TrueTargetInfo.flag.reads) & ~BlockInfo.flag.reads;
|
||||
BlockInfo.flag.kill &= FalseTargetInfo.flag.writes & ~(FalseTargetInfo.flag.reads) & ~BlockInfo.flag.reads;
|
||||
|
||||
// Flags that are written by the next blocks can be considered as written by this block, if not read
|
||||
BlockInfo.flag.writes |= BlockInfo.flag.kill & ~BlockInfo.flag.reads;
|
||||
|
||||
|
||||
//// GPRs ////
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
BlockInfo.gpr.kill = TrueTargetInfo.gpr.writes & ~(TrueTargetInfo.gpr.reads) & ~BlockInfo.gpr.reads;
|
||||
BlockInfo.gpr.kill &= FalseTargetInfo.gpr.writes & ~(FalseTargetInfo.gpr.reads) & ~BlockInfo.gpr.reads;
|
||||
|
||||
// GPRs that are written by the next blocks can be considered as written by this block, if not read
|
||||
BlockInfo.gpr.writes |= BlockInfo.gpr.kill & ~BlockInfo.gpr.reads;
|
||||
|
||||
|
||||
//// FPRs ////
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
BlockInfo.fpr.kill = TrueTargetInfo.fpr.writes & ~(TrueTargetInfo.fpr.reads) & ~BlockInfo.fpr.reads;
|
||||
BlockInfo.fpr.kill &= FalseTargetInfo.fpr.writes & ~(FalseTargetInfo.fpr.reads) & ~BlockInfo.fpr.reads;
|
||||
|
||||
// FPRs that are written by the next blocks can be considered as written by this block, if not read
|
||||
BlockInfo.fpr.writes |= BlockInfo.fpr.kill & ~BlockInfo.fpr.reads;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 3
|
||||
// Remove the dead stores
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
// If this StoreFlag is never read, remove it
|
||||
if (BlockInfo.flag.kill & (1UL << Op->Flag)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPRs ////
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if (BlockInfo.gpr.kill & GPRBit(Op->Offset)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
//// FPRs ////
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if ((BlockInfo.fpr.kill & FPRBit(Op->Offset, IROp->Size)) == FPRBit(Op->Offset, IROp->Size) && (FPRBit(Op->Offset, IROp->Size) != 0)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateDeadStoreElimination() {
|
||||
return new DeadStoreElimination{};
|
||||
}
|
||||
|
||||
}
|
||||
+24
-14
@@ -6,6 +6,13 @@
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
// struct to avoid zero-initialization
|
||||
struct RemapNode {
|
||||
IR::OrderedNodeWrapper::NodeOffsetType NodeID;
|
||||
};
|
||||
|
||||
static_assert(sizeof(RemapNode) == 4);
|
||||
|
||||
class IRCompaction final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
IRCompaction();
|
||||
@@ -14,7 +21,7 @@ public:
|
||||
private:
|
||||
static constexpr size_t AlignSize = 0x2000;
|
||||
OpDispatchBuilder LocalBuilder;
|
||||
std::vector<IR::OrderedNodeWrapper::NodeOffsetType> OldToNewRemap;
|
||||
std::vector<RemapNode> OldToNewRemap;
|
||||
struct CodeBlockData {
|
||||
OrderedNode *OldNode;
|
||||
OrderedNode *NewNode;
|
||||
@@ -35,7 +42,10 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
if (OldToNewRemap.size() < NodeCount) {
|
||||
OldToNewRemap.resize(std::max(OldToNewRemap.size() * 2U, AlignUp(NodeCount, AlignSize)));
|
||||
}
|
||||
memset(&OldToNewRemap.at(0), 0xFF, NodeCount * sizeof(IR::OrderedNodeWrapper::NodeOffsetType));
|
||||
#ifndef NDEBUG
|
||||
memset(&OldToNewRemap.at(0), 0xFF, NodeCount * sizeof(RemapNode));
|
||||
#endif
|
||||
|
||||
GeneratedCodeBlocks.clear();
|
||||
|
||||
// Reset our local working list
|
||||
@@ -46,7 +56,6 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
uintptr_t LocalDataBegin = LocalIR.GetData();
|
||||
|
||||
uintptr_t ListBegin = CurrentIR.GetListData();
|
||||
uintptr_t DataBegin = CurrentIR.GetData();
|
||||
|
||||
auto HeaderNode = CurrentIR.GetHeaderNode();
|
||||
auto HeaderOp = CurrentIR.GetHeader();
|
||||
@@ -67,9 +76,9 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
// Then create all the ops inside the code blocks
|
||||
|
||||
// Zero is always zero(invalid)
|
||||
OldToNewRemap[0] = 0;
|
||||
OldToNewRemap[0].NodeID = 0;
|
||||
auto LocalHeaderOp = LocalBuilder._IRHeader(OrderedNodeWrapper::WrapOffset(0).GetNode(ListBegin), HeaderOp->Entry, HeaderOp->BlockCount, HeaderOp->ShouldInterpret);
|
||||
OldToNewRemap[CurrentIR.GetID(HeaderNode)] = LocalIR.GetID(LocalHeaderOp.Node);
|
||||
OldToNewRemap[CurrentIR.GetID(HeaderNode)].NodeID = LocalIR.GetID(LocalHeaderOp.Node);
|
||||
|
||||
{
|
||||
// Generate our codeblocks and link them together
|
||||
@@ -77,7 +86,7 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
LogMan::Throw::A(BlockHeader->Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
auto LocalBlockIRNode = LocalBuilder._CodeBlock(LocalHeaderOp, LocalHeaderOp); // Use LocalHeaderOp as a dummy arg for now
|
||||
OldToNewRemap[CurrentIR.GetID(BlockNode)] = LocalIR.GetID(LocalBlockIRNode.Node);
|
||||
OldToNewRemap[CurrentIR.GetID(BlockNode)].NodeID = LocalIR.GetID(LocalBlockIRNode.Node);
|
||||
GeneratedCodeBlocks.emplace_back(CodeBlockData{BlockNode, LocalBlockIRNode});
|
||||
}
|
||||
|
||||
@@ -112,7 +121,7 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
// Set our map remapper to map the new location
|
||||
// Even nodes that don't have a destination need to be in this map
|
||||
// Need to be able to remap branch targets any other bits
|
||||
OldToNewRemap[CurrentIR.GetID(CodeNode)] = LocalIR.GetID(LocalPair.Node);
|
||||
OldToNewRemap[CurrentIR.GetID(CodeNode)].NodeID = LocalIR.GetID(LocalPair.Node);
|
||||
|
||||
if (i == 0) {
|
||||
FirstNode.OldNode = CodeNode;
|
||||
@@ -137,20 +146,21 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
{
|
||||
// Fixup the arguments of all the IROps
|
||||
for (auto &Block : GeneratedCodeBlocks) {
|
||||
auto BlockIROp = CurrentIR.GetOp<FEXCore::IR::IROp_CodeBlock>(Block.OldNode);
|
||||
auto BlockIROp = LocalIR.GetOp<FEXCore::IR::IROp_CodeBlock>(Block.NewNode);
|
||||
LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(Block.OldNode)) {
|
||||
auto [LocalNode, LocalIROp] = LocalIR.at(OldToNewRemap[CurrentIR.GetID(CodeNode)])();
|
||||
for (auto [LocalNode, LocalIROp] : LocalIR.GetCode(Block.NewNode)) {
|
||||
|
||||
// Now that we have the op copied over, we need to modify SSA values to point to the new correct locations
|
||||
// This doesn't use IR::GetArgs(Op) because we need to remap all SSA nodes
|
||||
// Including ones that we don't RA
|
||||
uint8_t NumArgs = IROp->NumArgs;
|
||||
uint8_t NumArgs = LocalIROp->NumArgs;
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
uint32_t OldArg = IROp->Args[i].ID();
|
||||
LogMan::Throw::A(OldToNewRemap[OldArg] != ~0U, "Tried remapping unfound node %%ssa%d", OldArg);
|
||||
LocalIROp->Args[i].NodeOffset = OldToNewRemap[OldArg] * sizeof(OrderedNode);
|
||||
uint32_t OldArg = LocalIROp->Args[i].ID();
|
||||
#ifndef NDEBUG
|
||||
LogMan::Throw::A(OldToNewRemap[OldArg].NodeID != ~0U, "Tried remapping unfound node %%ssa%d", OldArg);
|
||||
#endif
|
||||
LocalIROp->Args[i].NodeOffset = OldToNewRemap[OldArg].NodeID * sizeof(OrderedNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -50,9 +50,9 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto HeaderOp = CurrentIR.GetHeader();
|
||||
LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
|
||||
IR::RegisterAllocationPass * RAPass{};
|
||||
IR::RegisterAllocationData * RAData{};
|
||||
if (Manager->HasRAPass() && !HeaderOp->ShouldInterpret) {
|
||||
RAPass = Manager->GetRAPass();
|
||||
RAData = Manager->GetRAPass() ? Manager->GetRAPass()->GetAllocationData() : nullptr;
|
||||
}
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
@@ -81,12 +81,12 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
Warnings << "%ssa" << ID << ": Destination created but had no uses" << std::endl;
|
||||
}
|
||||
|
||||
if (RAPass) {
|
||||
if (RAData) {
|
||||
// If we have a register allocator then the destination needs to be assigned a register and class
|
||||
uint64_t Reg = RAPass->GetNodeRegister(ID);
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
|
||||
FEXCore::IR::RegisterClassType ExpectedClass = IR::GetRegClass(IROp->Op);
|
||||
FEXCore::IR::RegisterClassType AssignedClass = FEXCore::IR::RegisterClassType{uint32_t(Reg >> 32)};
|
||||
FEXCore::IR::RegisterClassType AssignedClass = FEXCore::IR::RegisterClassType{PhyReg.Class};
|
||||
|
||||
// If no register class was assigned
|
||||
if (AssignedClass == IR::InvalidClass) {
|
||||
@@ -95,7 +95,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
}
|
||||
|
||||
// If no physical register was assigned
|
||||
if ((uint32_t)Reg == ~0U) {
|
||||
if (PhyReg.Reg == IR::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Had destination but with no register assigned" << std::endl;
|
||||
}
|
||||
@@ -260,7 +260,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
HadWarning = false;
|
||||
if (HadError || HadWarning) {
|
||||
FEXCore::IR::Dump(&Out, &CurrentIR, RAPass);
|
||||
FEXCore::IR::Dump(&Out, &CurrentIR, RAData);
|
||||
|
||||
if (HadError) {
|
||||
Out << "Errors:" << std::endl << Errors.str() << std::endl;
|
||||
@@ -269,7 +269,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
if (HadWarning) {
|
||||
Out << "Warnings:" << std::endl << Warnings.str() << std::endl;
|
||||
}
|
||||
|
||||
|
||||
LogMan::Msg::E("%s", Out.str().c_str());
|
||||
}
|
||||
|
||||
|
||||
+459
-342
File diff suppressed because it is too large.
Load diff
@@ -1,5 +1,6 @@
|
||||
#pragma once
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
@@ -9,7 +10,6 @@ class IRListView;
|
||||
class RegisterAllocationPass : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool HasFullRA() const { return HadFullRA; }
|
||||
uint32_t SpillSlots() const { return SpillSlotCount; }
|
||||
|
||||
virtual void AllocateRegisterSet(uint32_t RegisterCount, uint32_t ClassCount) = 0;
|
||||
virtual void AddRegisters(FEXCore::IR::RegisterClassType Class, uint32_t RegisterCount) = 0;
|
||||
@@ -33,10 +33,14 @@ class RegisterAllocationPass : public FEXCore::IR::Pass {
|
||||
* @{ */
|
||||
|
||||
/**
|
||||
* @brief Returns the register and class encoded together
|
||||
* Top 32bits is the class, lower 32bits is the register
|
||||
* @brief Returns the register and class map array
|
||||
*/
|
||||
virtual uint64_t GetNodeRegister(uint32_t Node) = 0;
|
||||
virtual RegisterAllocationData *GetAllocationData() = 0;
|
||||
|
||||
/**
|
||||
* @brief Returns and transfers ownership of the register and class map array
|
||||
*/
|
||||
virtual std::unique_ptr<RegisterAllocationData, RegisterAllocationDataDeleter> PullAllocationData() = 0;
|
||||
/** @} */
|
||||
|
||||
protected:
|
||||
|
||||
+2
-2
@@ -50,7 +50,7 @@ bool StaticRegisterAllocationPass::Run(IREmitter *IREmit) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
|
||||
if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadContext>();
|
||||
|
||||
@@ -79,7 +79,7 @@ bool StaticRegisterAllocationPass::Run(IREmitter *IREmit) {
|
||||
}
|
||||
|
||||
auto StaticClass = GeneralClass == GPRClass ? GPRFixedClass : FPRFixedClass;
|
||||
OrderedNode *sraReg = IREmit->_StoreRegister(val, false, Op->Offset, GeneralClass, StaticClass, Op->Header.Size);
|
||||
IREmit->_StoreRegister(val, false, Op->Offset, GeneralClass, StaticClass, Op->Header.Size);
|
||||
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
|
||||
+14
-15
@@ -1,3 +1,4 @@
|
||||
#include <FEXCore/Utils/Common/MathUtils.h>
|
||||
#include <FEXCore/Utils/ELFLoader.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <cstring>
|
||||
@@ -201,6 +202,9 @@ bool ELFContainer::LoadELF_32() {
|
||||
|
||||
DynamicProgram = Header._32.e_type != ET_EXEC;
|
||||
|
||||
// Default BRK size
|
||||
BRKSize = 4096;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -239,6 +243,9 @@ bool ELFContainer::LoadELF_64() {
|
||||
|
||||
DynamicProgram = Header._64.e_type != ET_EXEC;
|
||||
|
||||
// Default BRK size
|
||||
BRKSize = 0x1000'0000;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -321,8 +328,8 @@ void ELFContainer::CalculateMemoryLayouts() {
|
||||
//
|
||||
// We need to ignore such empty sections, or we will mistakenly assume the elf starts at zero.
|
||||
if (hdr->p_memsz > 0) {
|
||||
MinPhysAddr = std::min(MinPhysAddr, hdr->p_paddr);
|
||||
MaxPhysAddr = std::max(MaxPhysAddr, hdr->p_paddr + hdr->p_memsz);
|
||||
MinPhysAddr = std::min(MinPhysAddr, static_cast<uint64_t>(hdr->p_paddr));
|
||||
MaxPhysAddr = std::max(MaxPhysAddr, static_cast<uint64_t>(hdr->p_paddr + hdr->p_memsz));
|
||||
}
|
||||
if (hdr->p_type == PT_TLS) {
|
||||
TLSHeader._64 = hdr;
|
||||
@@ -330,6 +337,11 @@ void ELFContainer::CalculateMemoryLayouts() {
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate BRK
|
||||
MaxPhysAddr = AlignUp(MaxPhysAddr, 4096);
|
||||
BRKBase = MaxPhysAddr;
|
||||
MaxPhysAddr += BRKSize;
|
||||
|
||||
PhysMemSize = MaxPhysAddr - MinPhysAddr;
|
||||
|
||||
MinPhysicalMemoryLocation = MinPhysAddr;
|
||||
@@ -674,8 +686,6 @@ void ELFContainer::PrintProgramHeaders() const {
|
||||
if (Mode == MODE_32BIT) {
|
||||
LogMan::Throw::A(Header._32.e_shstrndx < SectionHeaders.size(),
|
||||
"String index section is wrong index!");
|
||||
Elf32_Shdr const *StrHeader = SectionHeaders.at(Header._32.e_shstrndx)._32;
|
||||
char const *SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
for (uint32_t i = 0; i < ProgramHeaders.size(); ++i) {
|
||||
Elf32_Phdr const *hdr = ProgramHeaders.at(i)._32;
|
||||
LogMan::Msg::I("Type: %d", hdr->p_type);
|
||||
@@ -691,8 +701,6 @@ void ELFContainer::PrintProgramHeaders() const {
|
||||
else {
|
||||
LogMan::Throw::A(Header._64.e_shstrndx < SectionHeaders.size(),
|
||||
"String index section is wrong index!");
|
||||
Elf64_Shdr const *StrHeader = SectionHeaders.at(Header._64.e_shstrndx)._64;
|
||||
char const *SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
for (uint32_t i = 0; i < ProgramHeaders.size(); ++i) {
|
||||
Elf64_Phdr const *hdr = ProgramHeaders.at(i)._64;
|
||||
LogMan::Msg::I("Type: %d", hdr->p_type);
|
||||
@@ -884,9 +892,6 @@ void ELFContainer::FixupRelocations(void *ELFBase, uint64_t GuestELFBase, Symbol
|
||||
Elf64_Shdr const *GOTHeader {nullptr};
|
||||
Elf64_Shdr const *DynSymHeader {nullptr};
|
||||
|
||||
Elf64_Shdr const *StrHeader = SectionHeaders.at(Header._64.e_shstrndx)._64;
|
||||
char const *SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
|
||||
Elf64_Shdr const *StringTableHeader{nullptr};
|
||||
char const *StrTab{nullptr};
|
||||
|
||||
@@ -1206,9 +1211,6 @@ void ELFContainer::GetInitLocations(uint64_t GuestELFBase, std::vector<uint64_t>
|
||||
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
Elf32_Shdr const *hdr = SectionHeaders.at(i)._32;
|
||||
if (hdr->sh_type == SHT_DYNAMIC) {
|
||||
Elf32_Shdr const *StrHeader = SectionHeaders.at(hdr->sh_link)._32;
|
||||
char const *SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
|
||||
size_t Entries = hdr->sh_size / hdr->sh_entsize;
|
||||
for (size_t j = 0; i < Entries; ++j) {
|
||||
Elf32_Dyn const *Dynamic = reinterpret_cast<Elf32_Dyn const*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize));
|
||||
@@ -1236,9 +1238,6 @@ void ELFContainer::GetInitLocations(uint64_t GuestELFBase, std::vector<uint64_t>
|
||||
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
Elf64_Shdr const *hdr = SectionHeaders.at(i)._64;
|
||||
if (hdr->sh_type == SHT_DYNAMIC) {
|
||||
Elf64_Shdr const *StrHeader = SectionHeaders.at(hdr->sh_link)._64;
|
||||
char const *SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
|
||||
size_t Entries = hdr->sh_size / hdr->sh_entsize;
|
||||
for (size_t j = 0; i < Entries; ++j) {
|
||||
Elf64_Dyn const *Dynamic = reinterpret_cast<Elf64_Dyn const*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize));
|
||||
|
||||
@@ -3,7 +3,9 @@
|
||||
|
||||
#include <list>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <stdint.h>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace FEXCore::Config {
|
||||
enum ConfigOption {
|
||||
@@ -22,6 +24,7 @@ namespace FEXCore::Config {
|
||||
CONFIG_ABI_LOCAL_FLAGS,
|
||||
CONFIG_ABI_NO_PF,
|
||||
CONFIG_DUMPIR,
|
||||
CONFIG_VALIDATE_IR_PARSER,
|
||||
CONFIG_SILENTLOGS,
|
||||
CONFIG_ENVIRONMENT,
|
||||
CONFIG_OUTPUTLOG,
|
||||
|
||||
+2
-1
@@ -7,6 +7,7 @@ namespace FEXCore {
|
||||
namespace IR {
|
||||
template<bool Copy>
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
}
|
||||
|
||||
namespace Core {
|
||||
@@ -44,7 +45,7 @@ class LLVMCore;
|
||||
* @return An executable function pointer that is theoretically compiled from this point.
|
||||
* Is actually a function pointer of type `void (FEXCore::Core::ThreadState *Thread)
|
||||
*/
|
||||
virtual void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData) = 0;
|
||||
virtual void *CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) = 0;
|
||||
|
||||
/**
|
||||
* @brief Function for mapping memory in to the CPUBackend's visible space. Allows setting up virtual mappings if required
|
||||
|
||||
@@ -47,6 +47,11 @@ namespace FEXCore::Core {
|
||||
|
||||
static_assert(std::is_standard_layout<ThreadState>::value, "This needs to be standard layout");
|
||||
|
||||
#ifdef PAGE_SIZE
|
||||
static_assert(PAGE_SIZE == 4096, "FEX only supports 4k pages");
|
||||
#undef PAGE_SIZE
|
||||
#endif
|
||||
|
||||
constexpr uint64_t PAGE_SIZE = 4096;
|
||||
|
||||
std::string_view const& GetFlagName(unsigned Flag);
|
||||
|
||||
+6
-4
@@ -2,8 +2,6 @@
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
#include <bits/types/sigset_t.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace x86_64 {
|
||||
// uc_flags flags
|
||||
@@ -74,18 +72,22 @@ namespace FEXCore {
|
||||
};
|
||||
static_assert(sizeof(FEXCore::x86_64::mcontext_t) == 256, "This needs to be the right size");
|
||||
|
||||
struct __attribute__((packed)) sigset_t {
|
||||
uint64_t val[16];
|
||||
};
|
||||
static_assert(sizeof(FEXCore::x86_64::sigset_t) == 128, "This needs to be the right size");
|
||||
|
||||
struct __attribute__((packed)) ucontext_t {
|
||||
uint64_t uc_flags;
|
||||
FEXCore::x86_64::ucontext_t *uc_link;
|
||||
FEXCore::x86_64::stack_t uc_stack;
|
||||
FEXCore::x86_64::mcontext_t uc_mcontext;
|
||||
sigset_t uc_sigmask;
|
||||
FEXCore::x86_64::sigset_t uc_sigmask;
|
||||
FEXCore::x86_64::_libc_fpstate __fpregs_mem;
|
||||
uint64_t __ssp[4];
|
||||
};
|
||||
static_assert(offsetof(FEXCore::x86_64::ucontext_t, uc_mcontext) == 40, "Needs to be correct");
|
||||
|
||||
static_assert(sizeof(sigset_t) == 128, "This needs to be the right size");
|
||||
static_assert(sizeof(FEXCore::x86_64::ucontext_t) == 968, "This needs to be the right size");
|
||||
}
|
||||
|
||||
|
||||
@@ -3,12 +3,14 @@
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <map>
|
||||
|
||||
#include <unordered_map>
|
||||
#include <thread>
|
||||
|
||||
namespace FEXCore {
|
||||
class BlockCache;
|
||||
class LookupCache;
|
||||
class CompileService;
|
||||
}
|
||||
|
||||
@@ -71,14 +73,12 @@ namespace FEXCore::Core {
|
||||
|
||||
std::unique_ptr<FEXCore::IR::OpDispatchBuilder> OpDispatcher;
|
||||
|
||||
std::shared_ptr<FEXCore::CPU::CPUBackend> CPUBackend;
|
||||
std::shared_ptr<FEXCore::CPU::CPUBackend> IntBackend;
|
||||
std::unique_ptr<FEXCore::CPU::CPUBackend> FallbackBackend;
|
||||
|
||||
std::unique_ptr<FEXCore::BlockCache> BlockCache;
|
||||
std::unique_ptr<FEXCore::CPU::CPUBackend> CPUBackend;
|
||||
std::unique_ptr<FEXCore::LookupCache> LookupCache;
|
||||
|
||||
std::unordered_map<uint64_t, std::unique_ptr<FEXCore::IR::IRListView<true>>> IRLists;
|
||||
std::unordered_map<uint64_t, FEXCore::Core::DebugData> DebugData;
|
||||
std::unordered_map<uint64_t, std::unique_ptr<FEXCore::IR::RegisterAllocationData, FEXCore::IR::RegisterAllocationDataDeleter>> RALists;
|
||||
std::unordered_map<uint64_t, std::unique_ptr<FEXCore::Core::DebugData>> DebugData;
|
||||
|
||||
std::unique_ptr<FEXCore::Frontend::Decoder> FrontendDecoder;
|
||||
std::unique_ptr<FEXCore::IR::PassManager> PassManager;
|
||||
|
||||
+10
-7
@@ -7,6 +7,7 @@
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
/**
|
||||
* @brief The IROp_Header is an dynamically sized array
|
||||
@@ -287,29 +288,29 @@ struct MemOffsetType final {
|
||||
};
|
||||
|
||||
struct TypeDefinition final {
|
||||
uint8_t Val;
|
||||
operator uint8_t() const {
|
||||
uint16_t Val;
|
||||
operator uint16_t() const {
|
||||
return Val;
|
||||
}
|
||||
|
||||
static TypeDefinition Create(uint8_t Bytes) {
|
||||
TypeDefinition Type{};
|
||||
Type.Val = Bytes << 2;
|
||||
Type.Val = Bytes << 8;
|
||||
return Type;
|
||||
}
|
||||
|
||||
static TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
|
||||
TypeDefinition Type{};
|
||||
Type.Val = (Bytes << 2) | (Elements & 0b11);
|
||||
Type.Val = (Bytes << 8) | (Elements & 255);
|
||||
return Type;
|
||||
}
|
||||
|
||||
uint8_t Bytes() const {
|
||||
return Val >> 2;
|
||||
return Val >> 8;
|
||||
}
|
||||
|
||||
uint8_t Elements() const {
|
||||
return Val & 0b11;
|
||||
return Val & 255;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -457,8 +458,10 @@ public:
|
||||
|
||||
template<bool>
|
||||
class IRListView;
|
||||
class IREmitter;
|
||||
|
||||
void Dump(std::stringstream *out, IRListView<false> const* IR, IR::RegisterAllocationPass *RAPass);
|
||||
void Dump(std::stringstream *out, IRListView<false> const* IR, IR::RegisterAllocationData *RAData);
|
||||
IREmitter* Parse(std::istream *in);
|
||||
|
||||
template<typename Type>
|
||||
inline uint32_t NodeWrapperBase<Type>::ID() const { return NodeOffset / sizeof(IR::OrderedNode); }
|
||||
|
||||
+6
-2
@@ -40,7 +40,8 @@ friend class FEXCore::IR::PassManager;
|
||||
|
||||
IRPair<IROp_Constant> _Constant(uint8_t Size, uint64_t Constant) {
|
||||
auto Op = AllocateOp<IROp_Constant, IROps::OP_CONSTANT>();
|
||||
Op.first->Constant = Constant;
|
||||
uint64_t Mask = ~0ULL >> (Size - 64);
|
||||
Op.first->Constant = (Constant & Mask);
|
||||
Op.first->Header.Size = Size / 8;
|
||||
Op.first->Header.ElementSize = Size / 8;
|
||||
Op.first->Header.NumArgs = 0;
|
||||
@@ -504,9 +505,12 @@ friend class FEXCore::IR::PassManager;
|
||||
LogMan::Throw::A(rhs.ListData.BackingSize() <= ListData.BackingSize(), "Trying to take ownership of data that is too large");
|
||||
Data.CopyData(rhs.Data);
|
||||
ListData.CopyData(rhs.ListData);
|
||||
InvalidNode = rhs.InvalidNode;
|
||||
InvalidNode = rhs.InvalidNode->Wrapped(rhs.ListData.Begin()).GetNode(ListData.Begin());
|
||||
CurrentWriteCursor = rhs.CurrentWriteCursor;
|
||||
CodeBlocks = rhs.CodeBlocks;
|
||||
for (auto& CodeBlock: CodeBlocks) {
|
||||
CodeBlock = CodeBlock->Wrapped(rhs.ListData.Begin()).GetNode(ListData.Begin());
|
||||
}
|
||||
}
|
||||
|
||||
void SetWriteCursor(OrderedNode *Node) {
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
#pragma once
|
||||
#include "IR.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
union PhysicalRegister {
|
||||
uint8_t Raw;
|
||||
struct {
|
||||
uint8_t Reg: 5;
|
||||
uint8_t Class: 3;
|
||||
};
|
||||
|
||||
bool operator==(const PhysicalRegister &Other) const {
|
||||
return Raw == Other.Raw;
|
||||
}
|
||||
|
||||
PhysicalRegister(RegisterClassType Class, uint8_t Reg) : Reg(Reg), Class(Class.Val) { }
|
||||
|
||||
static const PhysicalRegister Invalid() {
|
||||
return PhysicalRegister(InvalidClass, InvalidReg);
|
||||
}
|
||||
|
||||
bool IsInvalid() {
|
||||
return *this == Invalid();
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(PhysicalRegister) == 1);
|
||||
|
||||
class RegisterAllocationData;
|
||||
struct RegisterAllocationDataDeleter {
|
||||
void operator()(RegisterAllocationData* r) {
|
||||
free(r);
|
||||
}
|
||||
};
|
||||
|
||||
class RegisterAllocationData {
|
||||
public:
|
||||
uint32_t SpillSlotCount {};
|
||||
PhysicalRegister Map[0];
|
||||
|
||||
PhysicalRegister GetNodeRegister(uint32_t Node) const {
|
||||
return Map[Node];
|
||||
}
|
||||
uint32_t SpillSlots() const { return SpillSlotCount; }
|
||||
|
||||
static size_t Size(uint32_t NodeCount) {
|
||||
return sizeof(RegisterAllocationData) + NodeCount * sizeof(Map[0]);
|
||||
}
|
||||
};
|
||||
|
||||
}
|
||||
@@ -8,6 +8,15 @@
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
// Add macros which are missing in some versions of <elf.h>
|
||||
#ifndef ELF32_ST_VISIBILITY
|
||||
#define ELF32_ST_VISIBILITY(o) ((o) & 0x3)
|
||||
#endif
|
||||
|
||||
#ifndef ELF64_ST_VISIBILITY
|
||||
#define ELF64_ST_VISIBILITY(o) ((o) & 0x3)
|
||||
#endif
|
||||
|
||||
namespace ELFLoader {
|
||||
struct ELFSymbol {
|
||||
uint64_t FileOffset;
|
||||
@@ -40,6 +49,15 @@ public:
|
||||
PhysicalMemorySize);
|
||||
}
|
||||
|
||||
struct BRKInfo {
|
||||
uint64_t Base;
|
||||
uint64_t Size;
|
||||
};
|
||||
|
||||
BRKInfo GetBRKInfo() const {
|
||||
return {BRKBase, BRKSize};
|
||||
}
|
||||
|
||||
// Data, Physical, Size
|
||||
using MemoryWriter = std::function<void(void *, uint64_t, uint64_t)>;
|
||||
void WriteLoadableSections(MemoryWriter Writer, uint64_t Offset = 0);
|
||||
@@ -67,6 +85,14 @@ public:
|
||||
void GetInitLocations(uint64_t GuestELFBase, std::vector<uint64_t> *Locations);
|
||||
|
||||
bool HasTLS() const { return TLSHeader._64 != nullptr; }
|
||||
uint64_t GetTLSBase() const {
|
||||
if (GetMode() == ELFMode::MODE_64BIT) {
|
||||
return TLSHeader._64->p_vaddr;
|
||||
}
|
||||
else {
|
||||
return TLSHeader._32->p_vaddr;
|
||||
}
|
||||
}
|
||||
|
||||
enum ELFMode {
|
||||
MODE_32BIT,
|
||||
@@ -122,6 +148,9 @@ private:
|
||||
uint64_t MinPhysicalMemoryLocation{0};
|
||||
uint64_t MaxPhysicalMemoryLocation{0};
|
||||
uint64_t PhysicalMemorySize{0};
|
||||
|
||||
uint64_t BRKBase{};
|
||||
uint64_t BRKSize{};
|
||||
ProgramHeader InterpreterHeader{};
|
||||
bool DynamicProgram{false};
|
||||
std::string DynamicLinker;
|
||||
|
||||
+1
@@ -1,3 +1,4 @@
|
||||
#pragma once
|
||||
|
||||
#define GIT_SHORT_HASH "@GIT_SHORT_HASH@"
|
||||
#define GIT_DESCRIBE_STRING "@GIT_DESCRIBE_STRING@"
|
||||
+1
Submodule External/fex-gcc-target-tests-bins added at 9f83d474bb.
@@ -7,7 +7,6 @@ This is the frontend application and tooling used for development and debugging
|
||||
* imgui
|
||||
* json-maker
|
||||
* tiny-json
|
||||
* boost interprocess (sadly)
|
||||
* A C++17 compliant compiler (There are assumptions made about using Clang and LTO)
|
||||
* clang-tidy if you want the code cleaned up
|
||||
* cmake
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
#!/usr/bin/python3
|
||||
import re
|
||||
import sys
|
||||
import subprocess
|
||||
|
||||
# Order this list from oldest to newest
|
||||
# try not to list something newer than our minimum compiler supported version
|
||||
BigCoreIDs = {
|
||||
# ARM
|
||||
tuple([0x41, 0xd07]): "cortex-a57",
|
||||
tuple([0x41, 0xd08]): "cortex-a72",
|
||||
tuple([0x41, 0xd09]): "cortex-a73",
|
||||
tuple([0x41, 0xd0a]): "cortex-a75",
|
||||
tuple([0x41, 0xd0b]): "cortex-a76",
|
||||
tuple([0x41, 0xd0d]): "cortex-a77",
|
||||
tuple([0x41, 0xd41]): "cortex-a78",
|
||||
tuple([0x41, 0xd44]): "cortex-x1",
|
||||
tuple([0x41, 0xd0c]): "neoverse-n1",
|
||||
tuple([0x41, 0xd49]): "neoverse-n2",
|
||||
## Nvidia
|
||||
tuple([0x4e, 0x004]): "carmel", # Carmel
|
||||
# Qualcomm
|
||||
tuple([0x51, 0x800]): "cortex-a73", # Kryo 2xx Gold
|
||||
tuple([0x51, 0x802]): "cortex-a75", # Kryo 3xx Gold
|
||||
tuple([0x51, 0x804]): "cortex-a76", # Kryo 4xx Gold
|
||||
}
|
||||
|
||||
LittleCoreIDs = {
|
||||
# ARM
|
||||
tuple([0x41, 0xd04]): "cortex-a35",
|
||||
tuple([0x41, 0xd03]): "cortex-a53",
|
||||
tuple([0x41, 0xd05]): "cortex-a55",
|
||||
|
||||
# Qualcomm
|
||||
tuple([0x51, 0x801]): "cortex-a53", # Kryo 2xx Silver
|
||||
tuple([0x51, 0x803]): "cortex-a55", # Kryo 3xx Silver
|
||||
tuple([0x51, 0x805]): "cortex-a55", # Kryo 4xx/5xx Silver
|
||||
}
|
||||
|
||||
# Args: </proc/cpuinfo file>
|
||||
if (len(sys.argv) < 2):
|
||||
sys.exit()
|
||||
|
||||
cpuinfo = []
|
||||
with open(sys.argv[1]) as cpuinfo_file:
|
||||
current_implementer = 0
|
||||
current_part = 0
|
||||
for line in cpuinfo_file:
|
||||
line = line.strip()
|
||||
if "CPU implementer" in line:
|
||||
current_implementer = int(re.findall(r'0x[0-9A-F]+', line, re.I)[0], 16)
|
||||
if "CPU part" in line:
|
||||
current_part = int(re.findall(r'0x[0-9A-F]+', line, re.I)[0], 16)
|
||||
cpuinfo += {tuple([current_implementer, current_part])}
|
||||
|
||||
largest_big = "native"
|
||||
largest_little = "native"
|
||||
|
||||
for core in cpuinfo:
|
||||
if BigCoreIDs.get(core):
|
||||
largest_big = BigCoreIDs.get(core)
|
||||
|
||||
if LittleCoreIDs.get(core):
|
||||
largest_little = LittleCoreIDs.get(core)
|
||||
|
||||
# We only want the big core output
|
||||
print(largest_big)
|
||||
# print(largest_little)
|
||||
@@ -58,8 +58,12 @@ else:
|
||||
Process.wait()
|
||||
ResultCode = Process.returncode
|
||||
|
||||
if (not test_name in expected_output or expected_output[test_name] != ResultCode):
|
||||
if expected_output.get(test_name):
|
||||
# expect zero by default
|
||||
if (not test_name in expected_output):
|
||||
expected_output[test_name] = 0
|
||||
|
||||
if (expected_output[test_name] != ResultCode):
|
||||
if (test_name in expected_output):
|
||||
print("test failed, expected is", expected_output[test_name], "but got", ResultCode)
|
||||
else:
|
||||
print("Test doesn't have expected output,", test_name)
|
||||
|
||||
@@ -23,7 +23,7 @@ namespace FEX::ArgLoader {
|
||||
#else
|
||||
.choices({"irint", "irjit"})
|
||||
#endif
|
||||
.set_default("irint");
|
||||
.set_default("irjit");
|
||||
|
||||
std::string BreakString = "Break";
|
||||
std::string MultiBlockString = "Multiblock";
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <cassert>
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
|
||||
@@ -30,7 +30,7 @@ namespace HostFactory {
|
||||
explicit HostCore(FEXCore::Context::Context* CTX, FEXCore::Core::ThreadState *Thread, bool Fallback);
|
||||
~HostCore() override;
|
||||
std::string GetName() override { return "Host Core"; }
|
||||
void* CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData) override;
|
||||
void* CompileCode(FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
void *MapRegion(void *HostPtr, uint64_t VirtualGuestPtr, uint64_t Size) override {
|
||||
return HostPtr;
|
||||
@@ -42,10 +42,6 @@ namespace HostFactory {
|
||||
bool HandleSIGSEGV(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context* CTX;
|
||||
FEXCore::Core::ThreadState *ThreadState;
|
||||
bool IsFallback{};
|
||||
|
||||
uint64_t ReturningStackLocation;
|
||||
uint64_t ThreadStopHandlerAddress;
|
||||
};
|
||||
@@ -54,10 +50,7 @@ namespace HostFactory {
|
||||
}
|
||||
|
||||
HostCore::HostCore(FEXCore::Context::Context* CTX, FEXCore::Core::ThreadState *Thread, bool Fallback)
|
||||
: CodeGenerator(4096)
|
||||
, CTX {CTX}
|
||||
, ThreadState {Thread}
|
||||
, IsFallback {Fallback} {
|
||||
: CodeGenerator(4096) {
|
||||
FEXCore::Context::RegisterHostSignalHandler(CTX, SIGSEGV,
|
||||
[](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
auto InternalThread = reinterpret_cast<FEXCore::Core::InternalThreadState*>(Thread);
|
||||
@@ -177,7 +170,7 @@ namespace HostFactory {
|
||||
ready();
|
||||
}
|
||||
|
||||
void* HostCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData) {
|
||||
void* HostCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
|
||||
+20
-93
@@ -1,81 +1,21 @@
|
||||
# Boost with minimum version of 1.50, not exact
|
||||
find_package(Boost 1.50 REQUIRED)
|
||||
|
||||
enable_language(ASM_NASM)
|
||||
if(NOT CMAKE_ASM_NASM_COMPILER_LOADED)
|
||||
error("Failed to find NASM compatible assembler!")
|
||||
endif()
|
||||
add_subdirectory(LinuxSyscalls)
|
||||
|
||||
set(SYSCALL_SRCS
|
||||
LinuxSyscalls/FileManagement.cpp
|
||||
LinuxSyscalls/EmulatedFiles/EmulatedFiles.cpp
|
||||
LinuxSyscalls/SignalDelegator.cpp
|
||||
LinuxSyscalls/Syscalls.cpp
|
||||
LinuxSyscalls/x32/Syscalls.cpp
|
||||
LinuxSyscalls/x32/EPoll.cpp
|
||||
LinuxSyscalls/x32/FD.cpp
|
||||
LinuxSyscalls/x32/FS.cpp
|
||||
LinuxSyscalls/x32/Info.cpp
|
||||
LinuxSyscalls/x32/Memory.cpp
|
||||
LinuxSyscalls/x32/NotImplemented.cpp
|
||||
LinuxSyscalls/x32/Semaphore.cpp
|
||||
LinuxSyscalls/x32/Sched.cpp
|
||||
LinuxSyscalls/x32/Signals.cpp
|
||||
LinuxSyscalls/x32/Socket.cpp
|
||||
LinuxSyscalls/x32/Thread.cpp
|
||||
LinuxSyscalls/x32/Time.cpp
|
||||
LinuxSyscalls/x32/Timer.cpp
|
||||
LinuxSyscalls/x64/EPoll.cpp
|
||||
LinuxSyscalls/x64/FD.cpp
|
||||
LinuxSyscalls/x64/IO.cpp
|
||||
LinuxSyscalls/x64/Ioctl.cpp
|
||||
LinuxSyscalls/x64/Info.cpp
|
||||
LinuxSyscalls/x64/Memory.cpp
|
||||
LinuxSyscalls/x64/Msg.cpp
|
||||
LinuxSyscalls/x64/NotImplemented.cpp
|
||||
LinuxSyscalls/x64/Semaphore.cpp
|
||||
LinuxSyscalls/x64/Sched.cpp
|
||||
LinuxSyscalls/x64/Signals.cpp
|
||||
LinuxSyscalls/x64/Socket.cpp
|
||||
LinuxSyscalls/x64/Thread.cpp
|
||||
LinuxSyscalls/x64/Syscalls.cpp
|
||||
LinuxSyscalls/x64/Time.cpp
|
||||
LinuxSyscalls/Syscalls/EPoll.cpp
|
||||
LinuxSyscalls/Syscalls/FD.cpp
|
||||
LinuxSyscalls/Syscalls/FS.cpp
|
||||
LinuxSyscalls/Syscalls/Info.cpp
|
||||
LinuxSyscalls/Syscalls/IO.cpp
|
||||
LinuxSyscalls/Syscalls/Key.cpp
|
||||
LinuxSyscalls/Syscalls/Memory.cpp
|
||||
LinuxSyscalls/Syscalls/Msg.cpp
|
||||
LinuxSyscalls/Syscalls/Sched.cpp
|
||||
LinuxSyscalls/Syscalls/Semaphore.cpp
|
||||
LinuxSyscalls/Syscalls/SHM.cpp
|
||||
LinuxSyscalls/Syscalls/Signals.cpp
|
||||
LinuxSyscalls/Syscalls/Socket.cpp
|
||||
LinuxSyscalls/Syscalls/Thread.cpp
|
||||
LinuxSyscalls/Syscalls/Time.cpp
|
||||
LinuxSyscalls/Syscalls/Timer.cpp
|
||||
LinuxSyscalls/Syscalls/NotImplemented.cpp
|
||||
LinuxSyscalls/Syscalls/Stubs.cpp
|
||||
)
|
||||
set(LIBS FEXCore Common CommonCore pthread)
|
||||
set(NAME FEXLoader)
|
||||
set(SRCS ELFLoader.cpp ${SYSCALL_SRCS})
|
||||
set(LIBS FEXCore Common CommonCore)
|
||||
|
||||
add_executable(${NAME} ${SRCS})
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
add_executable(FEXLoader ELFLoader.cpp)
|
||||
target_include_directories(FEXLoader PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
target_link_libraries(${NAME} ${LIBS})
|
||||
target_link_libraries(FEXLoader ${LIBS} LinuxEmulation)
|
||||
|
||||
install(TARGETS ${NAME}
|
||||
install(TARGETS FEXLoader
|
||||
RUNTIME
|
||||
DESTINATION bin
|
||||
COMPONENT runtime)
|
||||
|
||||
set(FEX_INTERP FEXInterpreter)
|
||||
install(CODE "
|
||||
EXECUTE_PROCESS(COMMAND ln -f ${NAME} ${FEX_INTERP}
|
||||
EXECUTE_PROCESS(COMMAND ln -f FEXLoader ${FEX_INTERP}
|
||||
WORKING_DIRECTORY ${CMAKE_INSTALL_PREFIX}/bin/
|
||||
)
|
||||
")
|
||||
@@ -115,39 +55,26 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
|
||||
endif()
|
||||
|
||||
set(NAME TestHarness)
|
||||
set(SRCS TestHarness.cpp)
|
||||
add_executable(TestHarness TestHarness.cpp)
|
||||
target_include_directories(TestHarness PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
add_executable(${NAME} ${SRCS})
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
target_link_libraries(TestHarness ${LIBS})
|
||||
|
||||
target_link_libraries(${NAME} ${LIBS})
|
||||
add_executable(TestHarnessRunner TestHarnessRunner.cpp)
|
||||
target_include_directories(TestHarnessRunner PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
set(NAME TestHarnessRunner)
|
||||
set(SRCS TestHarnessRunner.cpp
|
||||
${SYSCALL_SRCS})
|
||||
target_link_libraries(TestHarnessRunner ${LIBS} LinuxEmulation)
|
||||
|
||||
add_executable(${NAME} ${SRCS})
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
add_executable(UnitTestGenerator UnitTestGenerator.cpp)
|
||||
target_include_directories(UnitTestGenerator PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
target_link_libraries(${NAME} ${LIBS})
|
||||
target_link_libraries(UnitTestGenerator ${LIBS})
|
||||
|
||||
set(NAME UnitTestGenerator)
|
||||
set(SRCS UnitTestGenerator.cpp)
|
||||
|
||||
add_executable(${NAME} ${SRCS})
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
target_link_libraries(${NAME} ${LIBS})
|
||||
|
||||
set(NAME IRLoader)
|
||||
set(SRCS
|
||||
add_executable(IRLoader
|
||||
IRLoader.cpp
|
||||
IRLoader/Loader.cpp
|
||||
${SYSCALL_SRCS})
|
||||
)
|
||||
target_include_directories(IRLoader PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
add_executable(${NAME} ${SRCS})
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
target_link_libraries(${NAME} ${LIBS})
|
||||
target_link_libraries(IRLoader ${LIBS} LinuxEmulation)
|
||||
|
||||
@@ -191,10 +191,10 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS_INTERPRETER, IsInterpreter ? "1" : "0");
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_INTERPRETER_INSTALLED, IsInterpreterInstalled() ? "1" : "0");
|
||||
|
||||
FEXCore::Config::Value<uint8_t> CoreConfig{FEXCore::Config::CONFIG_DEFAULTCORE, 0};
|
||||
FEXCore::Config::Value<uint64_t> BlockSizeConfig{FEXCore::Config::CONFIG_MAXBLOCKINST, 1};
|
||||
FEXCore::Config::Value<uint8_t> CoreConfig{FEXCore::Config::CONFIG_DEFAULTCORE, 1};
|
||||
FEXCore::Config::Value<uint64_t> BlockSizeConfig{FEXCore::Config::CONFIG_MAXBLOCKINST, 5000};
|
||||
FEXCore::Config::Value<bool> SingleStepConfig{FEXCore::Config::CONFIG_SINGLESTEP, false};
|
||||
FEXCore::Config::Value<bool> MultiblockConfig{FEXCore::Config::CONFIG_MULTIBLOCK, false};
|
||||
FEXCore::Config::Value<bool> MultiblockConfig{FEXCore::Config::CONFIG_MULTIBLOCK, true};
|
||||
FEXCore::Config::Value<bool> GdbServerConfig{FEXCore::Config::CONFIG_GDBSERVER, false};
|
||||
FEXCore::Config::Value<std::string> LDPath{FEXCore::Config::CONFIG_ROOTFSPATH, ""};
|
||||
FEXCore::Config::Value<std::string> ThunkLibsPath{FEXCore::Config::CONFIG_THUNKLIBSPATH, ""};
|
||||
@@ -207,7 +207,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Config::Value<bool> ABILocalFlags{FEXCore::Config::CONFIG_ABI_LOCAL_FLAGS, false};
|
||||
FEXCore::Config::Value<bool> AbiNoPF{FEXCore::Config::CONFIG_ABI_NO_PF, false};
|
||||
|
||||
|
||||
::SilentLog = SilentLog();
|
||||
|
||||
if (!::SilentLog) {
|
||||
@@ -269,6 +268,8 @@ int main(int argc, char **argv, char **const envp) {
|
||||
CTX,
|
||||
SignalDelegation.get(),
|
||||
&Loader)};
|
||||
auto BRKInfo = Loader.GetBRKInfo();
|
||||
SyscallHandler->DefaultProgramBreak(BRKInfo.Base, BRKInfo.Size);
|
||||
|
||||
FEXCore::Context::SetSignalDelegator(CTX, SignalDelegation.get());
|
||||
FEXCore::Context::SetSyscallHandler(CTX, SyscallHandler.get());
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include "Common/MathUtils.h"
|
||||
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
#include <array>
|
||||
#include <bitset>
|
||||
#include <cassert>
|
||||
#include <cstring>
|
||||
@@ -327,6 +328,11 @@ namespace FEX::HarnessHelper {
|
||||
ConfigStructBase BaseConfig;
|
||||
};
|
||||
|
||||
#ifdef PAGE_SIZE
|
||||
static_assert(PAGE_SIZE == 4096, "FEX only supports 4k pages");
|
||||
#undef PAGE_SIZE
|
||||
#endif
|
||||
|
||||
class HarnessCodeLoader final : public FEXCore::CodeLoader {
|
||||
|
||||
static constexpr uint32_t PAGE_SIZE = 4096;
|
||||
@@ -495,13 +501,13 @@ public:
|
||||
//AuxVariables.emplace_back(auxv_t{24, ~0ULL}); // AT_PLATFORM
|
||||
// On x86 only allows userspace to check for monitor and fs/gs base writing in CPL3
|
||||
//AuxVariables.emplace_back(auxv_t{26, 0}); // AT_HWCAP2
|
||||
AuxVariables.emplace_back(auxv_t{32, 0ULL}); // sysinfo (vDSO)
|
||||
AuxVariables.emplace_back(auxv_t{33, 0ULL}); // sysinfo (vDSO)
|
||||
AuxVariables.emplace_back(auxv_t{32, 0}); // AT_SYSINFO - Entry point to syscall
|
||||
AuxVariables.emplace_back(auxv_t{33, 0}); // AT_SYSINFO_EHDR - Address of the start of VDSO
|
||||
}
|
||||
else {
|
||||
AuxVariables.emplace_back(auxv_t{4, 0x20}); // AT_PHENT
|
||||
AuxVariables.emplace_back(auxv_t{32, 0ULL}); // sysinfo (vDSO)
|
||||
AuxVariables.emplace_back(auxv_t{33, 0ULL}); // sysinfo (vDSO)
|
||||
AuxVariables.emplace_back(auxv_t{32, 0}); // AT_SYSINFO - Entry point to syscall
|
||||
AuxVariables.emplace_back(auxv_t{33, 0}); // AT_SYSINFO_EHDR - Address of the start of VDSO
|
||||
}
|
||||
|
||||
AuxVariables.emplace_back(auxv_t{3, DB.GetElfBase()}); // Program header
|
||||
@@ -554,8 +560,10 @@ public:
|
||||
size_t ArgSize = Args[i].size();
|
||||
// Set the pointer to this argument
|
||||
ArgumentPointers[i] = ArgumentBackingBaseGuest + CurrentOffset;
|
||||
// Copy the string in to the final location
|
||||
memcpy(reinterpret_cast<void*>(ArgumentBackingBase + CurrentOffset), &Args[i].at(0), ArgSize);
|
||||
if (ArgSize > 0) {
|
||||
// Copy the string in to the final location
|
||||
memcpy(reinterpret_cast<void*>(ArgumentBackingBase + CurrentOffset), &Args[i].at(0), ArgSize);
|
||||
}
|
||||
|
||||
// Set the null terminator for the string
|
||||
*reinterpret_cast<uint8_t*>(ArgumentBackingBase + CurrentOffset + ArgSize + 1) = 0;
|
||||
@@ -733,9 +741,16 @@ public:
|
||||
|
||||
bool Is64BitMode() const { return File.GetMode() == ::ELFLoader::ELFContainer::MODE_64BIT; }
|
||||
|
||||
::ELFLoader::ELFContainer::BRKInfo GetBRKInfo() const {
|
||||
auto Info = File.GetBRKInfo();
|
||||
Info.Base += DB.GetElfBase();
|
||||
return Info;
|
||||
}
|
||||
|
||||
private:
|
||||
::ELFLoader::ELFContainer File;
|
||||
::ELFLoader::ELFSymbolDatabase DB;
|
||||
|
||||
std::vector<std::string> Args;
|
||||
std::vector<std::string> EnvironmentVariables;
|
||||
std::vector<char const*> LoaderArgs;
|
||||
|
||||
@@ -81,7 +81,7 @@ class IRCodeLoader final : public FEXCore::CodeLoader {
|
||||
uint64_t GetFinalRIP() override { return 0; }
|
||||
|
||||
virtual void AddIR(IRHandler Handler) override {
|
||||
Handler(IR->GetEntryRIP(), IR);
|
||||
Handler(IR->GetEntryRIP(), IR->GetIREmitter());
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -124,7 +124,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
|
||||
FEXCore::Context::SetSignalDelegator(CTX, SignalDelegation.get());
|
||||
|
||||
FEX::IRLoader::InitializeStaticTables();
|
||||
FEX::IRLoader::Loader Loader(Args[0], Args[1]);
|
||||
|
||||
int Return{};
|
||||
|
||||
@@ -1,181 +1,12 @@
|
||||
#include "IRLoader/Loader.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
#include <fstream>
|
||||
|
||||
namespace {
|
||||
std::string ltrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(0, pos);
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
std::string rtrim(std::string String) {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(" \t\n\r")) != std::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
std::string trim(std::string String) {
|
||||
return rtrim(ltrim(String));
|
||||
}
|
||||
|
||||
std::string DecodeErrorToString(DecodeFailure Failure) {
|
||||
switch (Failure) {
|
||||
case DecodeFailure::DECODE_OKAY: return "Okay";
|
||||
case DecodeFailure::DECODE_UNKNOWN_TYPE: return "Unknown Type";
|
||||
case DecodeFailure::DECODE_INVALID: return "Invalid";
|
||||
case DecodeFailure::DECODE_INVALIDCHAR: return "Invalid starting char";
|
||||
case DecodeFailure::DECODE_INVALIDRANGE: return "Invalid integer range";
|
||||
case DecodeFailure::DECODE_INVALIDREGISTERCLASS: return "Invalid register class";
|
||||
case DecodeFailure::DECODE_UNKNOWN_SSA: return "Unknown SSA value";
|
||||
case DecodeFailure::DECODE_INVALID_CONDFLAG: return "Invalid Conditional name";
|
||||
case DecodeFailure::DECODE_INVALID_MEMOFFSETTYPE: return "Invalid Memory Offset Type";
|
||||
};
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
namespace FEX::IRLoader {
|
||||
std::unordered_map<std::string_view, FEXCore::IR::IROps> NameToOpMap;
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint8_t> Loader::DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint8_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint16_t> Loader::DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint16_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint32_t> Loader::DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint32_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint64_t> Loader::DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint64_t Result = strtoull(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::RegisterClassType> Loader::DecodeValue(std::string &Arg) {
|
||||
if (Arg == "GPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRClass};
|
||||
}
|
||||
else if (Arg == "FPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRClass};
|
||||
}
|
||||
else if (Arg == "GPRPair") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRPairClass};
|
||||
}
|
||||
else if (Arg == "Complex") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::ComplexClass};
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_INVALIDREGISTERCLASS, FEXCore::IR::InvalidClass};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::TypeDefinition> Loader::DecodeValue(std::string &Arg) {
|
||||
uint8_t Size{}, Elements{1};
|
||||
int NumArgs = sscanf(Arg.c_str(), "i%hhdv%hhd", &Size, &Elements);
|
||||
|
||||
if (NumArgs != 1 && NumArgs != 2) {
|
||||
return {DecodeFailure::DECODE_INVALID, {}};
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::TypeDefinition::Create(Size / 8, Elements)};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::CondClassType> Loader::DecodeValue(std::string &Arg) {
|
||||
std::array<std::string, 14> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
"UGE",
|
||||
"ULT",
|
||||
"MI",
|
||||
"PL",
|
||||
"VS",
|
||||
"VC",
|
||||
"UGT",
|
||||
"ULE",
|
||||
"SGE",
|
||||
"SLT",
|
||||
"SGT",
|
||||
"SLE",
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < CondNames.size(); ++i) {
|
||||
if (CondNames[i] == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, CondClassType{static_cast<uint8_t>(i)}};
|
||||
}
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_CONDFLAG, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::MemOffsetType> Loader::DecodeValue(std::string &Arg) {
|
||||
std::array<std::string, 3> Names = {
|
||||
"SXTX",
|
||||
"UXTW",
|
||||
"SXTW",
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < Names.size(); ++i) {
|
||||
if (Names[i] == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, MemOffsetType{static_cast<uint8_t>(i)}};
|
||||
}
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_MEMOFFSETTYPE, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, OrderedNode*> Loader::DecodeValue(std::string &Arg) {
|
||||
if (Arg.at(0) != '%') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
// Strip off the type qualifier from the ssa value
|
||||
size_t ArgEnd = std::string::npos;
|
||||
std::string SSAName = trim(Arg);
|
||||
ArgEnd = SSAName.find_first_of(" ");
|
||||
|
||||
if (ArgEnd != std::string::npos) {
|
||||
SSAName = SSAName.substr(0, ArgEnd);
|
||||
}
|
||||
|
||||
// Forward declarations may make this not succed
|
||||
auto Op = SSANameMapper.find(SSAName);
|
||||
if (Op == SSANameMapper.end()) {
|
||||
return {DecodeFailure::DECODE_UNKNOWN_SSA, nullptr};
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_OKAY, Op->second};
|
||||
}
|
||||
|
||||
Loader::Loader(std::string const &Filename, std::string const &ConfigFilename) {
|
||||
Config.Init(ConfigFilename);
|
||||
std::fstream fp(Filename, std::fstream::binary | std::fstream::in);
|
||||
@@ -185,324 +16,15 @@ namespace FEX::IRLoader {
|
||||
return;
|
||||
}
|
||||
|
||||
std::string TmpLine;
|
||||
while (!fp.eof()) {
|
||||
std::getline(fp, TmpLine);
|
||||
if (fp.eof()) {
|
||||
break;
|
||||
}
|
||||
if (fp.fail()) {
|
||||
LogMan::Msg::E("Failed to getline on line: %ld", Lines.size());
|
||||
return;
|
||||
}
|
||||
Lines.emplace_back(TmpLine);
|
||||
}
|
||||
ParsedCode.reset(FEXCore::IR::Parse(&fp));
|
||||
|
||||
ResetWorkingList();
|
||||
Loaded = Parse();
|
||||
}
|
||||
|
||||
bool Loader::Parse() {
|
||||
auto CheckPrintError = [&](LineDefinition &Def, DecodeFailure Failure) -> bool {
|
||||
if (Failure != DecodeFailure::DECODE_OKAY) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("Value Couldn't be decoded due to %s", DecodeErrorToString(Failure).c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
// String parse every line for our definitions
|
||||
for (size_t i = 0; i < Lines.size(); ++i) {
|
||||
std::string Line = Lines[i];
|
||||
LineDefinition Def{};
|
||||
CurrentDef = &Def;
|
||||
Def.LineNumber = i;
|
||||
|
||||
Line = trim(Line);
|
||||
|
||||
// Skip empty lines
|
||||
if (Line.empty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Line[0] == ';') {
|
||||
// This is a comment line
|
||||
// Skip it
|
||||
continue;
|
||||
}
|
||||
|
||||
size_t CurrentPos{};
|
||||
// Let's see if this node is assigning something first
|
||||
if (Line[0] == '%') {
|
||||
size_t DefinitionEnd = std::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of("=", CurrentPos)) != std::string::npos) {
|
||||
Def.Definition = Line.substr(0, DefinitionEnd);
|
||||
Def.Definition = trim(Def.Definition);
|
||||
Def.HasDefinition = true;
|
||||
CurrentPos = DefinitionEnd + 1; // +1 to ensure we go past then assignment
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("SSA declaration without assignment");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we are pulling in some IR from the IR Printer
|
||||
// Prints (%ssa%d) at the start of lines without a definition
|
||||
if (Line[0] == '(') {
|
||||
size_t DefinitionEnd = std::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of(")", CurrentPos)) != std::string::npos) {
|
||||
size_t SSAEnd = std::string::npos;
|
||||
if ((SSAEnd = Line.find_last_of(" ", DefinitionEnd)) != std::string::npos) {
|
||||
std::string Type = Line.substr(SSAEnd + 1, DefinitionEnd - SSAEnd - 1);
|
||||
Type = trim(Type);
|
||||
|
||||
auto DefinitionSize = DecodeValue<FEXCore::IR::TypeDefinition>(Type);
|
||||
if (!CheckPrintError(Def, DefinitionSize.first)) return false;
|
||||
Def.Size = DefinitionSize.second;
|
||||
}
|
||||
|
||||
Def.Definition = trim(Line.substr(1, std::min(DefinitionEnd, SSAEnd) - 1));
|
||||
|
||||
CurrentPos = DefinitionEnd + 1;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("SSA value with numbered SSA provided but no closing parentheses");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (Def.HasDefinition) {
|
||||
// Let's check if we have a size declared with this variable
|
||||
size_t NameEnd = std::string::npos;
|
||||
if ((NameEnd = Def.Definition.find_first_of(" ")) != std::string::npos) {
|
||||
std::string Type = Def.Definition.substr(NameEnd + 1);
|
||||
Type = trim(Type);
|
||||
Def.Definition = trim(Def.Definition.substr(0, NameEnd));
|
||||
|
||||
auto DefinitionSize = DecodeValue<FEXCore::IR::TypeDefinition>(Type);
|
||||
if (!CheckPrintError(Def, DefinitionSize.first)) return false;
|
||||
Def.Size = DefinitionSize.second;
|
||||
}
|
||||
|
||||
if (Def.Definition == "%Invalid") {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("Definition tried to define reserved %Invalid ssa node");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Let's get the IR op
|
||||
size_t OpNameEnd = std::string::npos;
|
||||
std::string RemainingLine = trim(Line.substr(CurrentPos));
|
||||
CurrentPos = 0;
|
||||
if ((OpNameEnd = RemainingLine.find_first_of(" \t\n\r\0", CurrentPos)) != std::string::npos) {
|
||||
Def.IROp = RemainingLine.substr(CurrentPos, OpNameEnd);
|
||||
Def.IROp = trim(Def.IROp);
|
||||
Def.HasArgs = true;
|
||||
CurrentPos = OpNameEnd;
|
||||
}
|
||||
else {
|
||||
if (RemainingLine.empty()) {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("Line without an IROp?");
|
||||
return false;
|
||||
}
|
||||
|
||||
Def.IROp = RemainingLine;
|
||||
Def.HasArgs = false;
|
||||
}
|
||||
|
||||
if (Def.HasArgs) {
|
||||
RemainingLine = trim(RemainingLine.substr(CurrentPos));
|
||||
CurrentPos = 0;
|
||||
if (RemainingLine.empty()) {
|
||||
// How did we get here?
|
||||
Def.HasArgs = false;
|
||||
}
|
||||
else {
|
||||
while (!RemainingLine.empty()) {
|
||||
size_t ArgEnd = std::string::npos;
|
||||
ArgEnd = RemainingLine.find_first_of(",");
|
||||
|
||||
std::string Arg = RemainingLine.substr(0, ArgEnd);
|
||||
Arg = trim(Arg);
|
||||
Def.Args.emplace_back(Arg);
|
||||
|
||||
RemainingLine.erase(0, ArgEnd+1); // +1 to ensure we go past the ','
|
||||
if (ArgEnd == std::string::npos)
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Defs.emplace_back(Def);
|
||||
}
|
||||
|
||||
// Ensure all of the ops are real ops
|
||||
for(size_t i = 0; i < Defs.size(); ++i) {
|
||||
auto &Def = Defs[i];
|
||||
auto Op = NameToOpMap.find(Def.IROp);
|
||||
if (Op == NameToOpMap.end()) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("IROp '%s' doesn't exist", Def.IROp.c_str());
|
||||
return false;
|
||||
}
|
||||
Def.OpEnum = Op->second;
|
||||
}
|
||||
|
||||
// Emit the header op
|
||||
IRPair<IROp_IRHeader> IRHeader;
|
||||
{
|
||||
auto &Def = Defs[0];
|
||||
CurrentDef = &Def;
|
||||
if (Def.OpEnum != FEXCore::IR::IROps::OP_IRHEADER) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("First op needs to be IRHeader. Was '%s'", Def.IROp.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Entry = DecodeValue<uint64_t>(Def.Args[0]);
|
||||
auto CodeBlockCount = DecodeValue<uint64_t>(Def.Args[2]);
|
||||
|
||||
if (!CheckPrintError(Def, Entry.first)) return false;
|
||||
if (!CheckPrintError(Def, CodeBlockCount.first)) return false;
|
||||
|
||||
EntryRIP = Entry.second;
|
||||
IRHeader = _IRHeader(InvalidNode, Entry.second, CodeBlockCount.second, false);
|
||||
}
|
||||
|
||||
SetWriteCursor(nullptr); // isolate the header from everything following
|
||||
|
||||
// Initialize SSANameMapper with Invalid value
|
||||
SSANameMapper["%Invalid"] = Invalid();
|
||||
|
||||
// Spin through the blocks and generate basic block ops
|
||||
for(size_t i = 0; i < Defs.size(); ++i) {
|
||||
auto &Def = Defs[i];
|
||||
if (Def.OpEnum == FEXCore::IR::IROps::OP_CODEBLOCK) {
|
||||
auto CodeBlock = _CodeBlock(InvalidNode, InvalidNode);
|
||||
SSANameMapper[Def.Definition] = CodeBlock.Node;
|
||||
Def.Node = CodeBlock.Node;
|
||||
|
||||
if (i == 1) {
|
||||
// First code block is the entry block
|
||||
// Link the header to the first block
|
||||
IRHeader.first->Blocks = CodeBlock.Node->Wrapped(ListData.Begin());
|
||||
}
|
||||
}
|
||||
}
|
||||
SetWriteCursor(nullptr); // isolate the block headers too
|
||||
|
||||
// Spin through all the definitions and add the ops to the basic blocks
|
||||
OrderedNode *CurrentBlock{};
|
||||
FEXCore::IR::IROp_CodeBlock *CurrentBlockOp{};
|
||||
for(size_t i = 1; i < Defs.size(); ++i) {
|
||||
auto &Def = Defs[i];
|
||||
CurrentDef = &Def;
|
||||
|
||||
if (Def.OpEnum == FEXCore::IR::IROps::OP_CODEBLOCK) {
|
||||
auto DefTarget = SSANameMapper.find(Def.Definition);
|
||||
if (CurrentBlock != nullptr) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("CodeBlock being used inside of already existing codeblock!");
|
||||
return false;
|
||||
}
|
||||
|
||||
CurrentBlock = DefTarget->second;
|
||||
CurrentBlockOp = CurrentBlock->Op(Data.Begin())->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
}
|
||||
|
||||
if (Def.OpEnum == FEXCore::IR::IROps::OP_ENDBLOCK) {
|
||||
if (CurrentBlock == nullptr) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("EndBlock being used outside of a block!");
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Adjust = DecodeValue<uint64_t>(Def.Args[0]);
|
||||
|
||||
if (!CheckPrintError(Def, Adjust.first)) return false;
|
||||
|
||||
Def.Node = _EndBlock(CurrentBlock);
|
||||
CurrentBlockOp->Last = Def.Node->Wrapped(ListData.Begin());
|
||||
|
||||
CurrentBlock = nullptr;
|
||||
CurrentBlockOp = nullptr;
|
||||
}
|
||||
|
||||
if (Def.OpEnum == FEXCore::IR::IROps::OP_DUMMY) {
|
||||
auto &PrevDef = Defs[i - 1];
|
||||
if (PrevDef.OpEnum != FEXCore::IR::IROps::OP_CODEBLOCK) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("Dummy op must be first op in block");
|
||||
return false;
|
||||
}
|
||||
|
||||
Def.Node = _Dummy();
|
||||
CurrentBlockOp->Begin = Def.Node->Wrapped(ListData.Begin());
|
||||
SSANameMapper[Def.Definition] = Def.Node;
|
||||
}
|
||||
|
||||
switch (Def.OpEnum) {
|
||||
// Special handled
|
||||
case FEXCore::IR::IROps::OP_IRHEADER:
|
||||
case FEXCore::IR::IROps::OP_CODEBLOCK:
|
||||
case FEXCore::IR::IROps::OP_ENDBLOCK:
|
||||
case FEXCore::IR::IROps::OP_DUMMY:
|
||||
break;
|
||||
#define IROP_PARSER_SWITCH_HELPERS
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
default: {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("Unhandled Op enum '%s' in parser", Def.IROp.c_str());
|
||||
return false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Def.HasDefinition) {
|
||||
auto IROp = Def.Node->Op(Data.Begin());
|
||||
if (Def.Size.Elements()) {
|
||||
IROp->Size = Def.Size.Bytes() * Def.Size.Elements();
|
||||
IROp->ElementSize = Def.Size.Bytes();
|
||||
}
|
||||
else {
|
||||
IROp->Size = Def.Size.Bytes();
|
||||
IROp->ElementSize = 0;
|
||||
}
|
||||
SSANameMapper[Def.Definition] = Def.Node;
|
||||
}
|
||||
}
|
||||
|
||||
std::stringstream out;
|
||||
auto NewIR = ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, nullptr);
|
||||
printf("IR:\n%s\n@@@@@\n", out.str().c_str());
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void InitializeStaticTables() {
|
||||
for (FEXCore::IR::IROps Op = FEXCore::IR::IROps::OP_DUMMY;
|
||||
Op <= FEXCore::IR::IROps::OP_LAST;
|
||||
Op = static_cast<FEXCore::IR::IROps>(static_cast<uint32_t>(Op) + 1)) {
|
||||
NameToOpMap[FEXCore::IR::GetName(Op)] = Op;
|
||||
if (ParsedCode) {
|
||||
auto NewIR = ParsedCode->ViewIR();
|
||||
EntryRIP = NewIR.GetHeader()->Entry;
|
||||
|
||||
std::stringstream out;
|
||||
FEXCore::IR::Dump(&out, &NewIR, nullptr);
|
||||
printf("IR:\n%s\n@@@@@\n", out.str().c_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -12,28 +12,13 @@
|
||||
|
||||
using namespace FEXCore::IR;
|
||||
|
||||
namespace {
|
||||
|
||||
enum class DecodeFailure {
|
||||
DECODE_OKAY,
|
||||
DECODE_UNKNOWN_TYPE,
|
||||
DECODE_INVALID,
|
||||
DECODE_INVALIDCHAR,
|
||||
DECODE_INVALIDRANGE,
|
||||
DECODE_INVALIDREGISTERCLASS,
|
||||
DECODE_UNKNOWN_SSA,
|
||||
DECODE_INVALID_CONDFLAG,
|
||||
DECODE_INVALID_MEMOFFSETTYPE,
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
namespace FEX::IRLoader {
|
||||
class Loader final : public FEXCore::IR::IREmitter {
|
||||
class Loader final {
|
||||
public:
|
||||
Loader(std::string const &Filename, std::string const &ConfigFilename);
|
||||
|
||||
bool IsValid() const { return Loaded; }
|
||||
bool IsValid() const { return ParsedCode != nullptr; }
|
||||
IREmitter* GetIREmitter() { return ParsedCode.get(); }
|
||||
uint64_t GetEntryRIP() const { return EntryRIP; }
|
||||
|
||||
bool CompareStates(FEXCore::Core::CPUState const* State) {
|
||||
@@ -50,74 +35,10 @@ namespace FEX::IRLoader {
|
||||
}
|
||||
}
|
||||
|
||||
#define IROP_PARSER_ALLOCATE_HELPERS
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
private:
|
||||
bool Parse();
|
||||
|
||||
uint64_t EntryRIP{};
|
||||
std::vector<std::string> Lines;
|
||||
bool Loaded{};
|
||||
std::unique_ptr<IREmitter> ParsedCode;
|
||||
|
||||
std::unordered_map<std::string, OrderedNode*> SSANameMapper;
|
||||
|
||||
template<typename Type>
|
||||
std::pair<DecodeFailure, Type> DecodeValue(std::string &Arg) {
|
||||
return {DecodeFailure::DECODE_UNKNOWN_TYPE, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint8_t>
|
||||
DecodeValue(std::string &Arg);
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint16_t>
|
||||
DecodeValue(std::string &Arg);
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint32_t>
|
||||
DecodeValue(std::string &Arg);
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint64_t>
|
||||
DecodeValue(std::string &Arg);
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::RegisterClassType>
|
||||
DecodeValue(std::string &Arg);
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::TypeDefinition>
|
||||
DecodeValue(std::string &Arg);
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::CondClassType>
|
||||
DecodeValue(std::string &Arg);
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::MemOffsetType>
|
||||
DecodeValue(std::string &Arg);
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, OrderedNode*>
|
||||
DecodeValue(std::string &Arg);
|
||||
|
||||
struct LineDefinition {
|
||||
size_t LineNumber;
|
||||
bool HasDefinition{};
|
||||
std::string Definition{};
|
||||
FEXCore::IR::TypeDefinition Size{};
|
||||
std::string IROp{};
|
||||
FEXCore::IR::IROps OpEnum;
|
||||
bool HasArgs{};
|
||||
std::vector<std::string> Args;
|
||||
OrderedNode *Node{};
|
||||
};
|
||||
|
||||
std::vector<LineDefinition> Defs;
|
||||
LineDefinition *CurrentDef{};
|
||||
FEX::HarnessHelper::ConfigLoader Config;
|
||||
};
|
||||
|
||||
void InitializeStaticTables();
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
|
||||
add_library(LinuxEmulation STATIC
|
||||
FileManagement.cpp
|
||||
EmulatedFiles/EmulatedFiles.cpp
|
||||
SignalDelegator.cpp
|
||||
Syscalls.cpp
|
||||
x32/Syscalls.cpp
|
||||
x32/EPoll.cpp
|
||||
x32/FD.cpp
|
||||
x32/FS.cpp
|
||||
x32/Info.cpp
|
||||
x32/Memory.cpp
|
||||
x32/NotImplemented.cpp
|
||||
x32/Semaphore.cpp
|
||||
x32/Sched.cpp
|
||||
x32/Signals.cpp
|
||||
x32/Socket.cpp
|
||||
x32/Thread.cpp
|
||||
x32/Time.cpp
|
||||
x32/Timer.cpp
|
||||
x64/EPoll.cpp
|
||||
x64/FD.cpp
|
||||
x64/IO.cpp
|
||||
x64/Ioctl.cpp
|
||||
x64/Info.cpp
|
||||
x64/Memory.cpp
|
||||
x64/Msg.cpp
|
||||
x64/NotImplemented.cpp
|
||||
x64/Semaphore.cpp
|
||||
x64/Sched.cpp
|
||||
x64/Signals.cpp
|
||||
x64/Socket.cpp
|
||||
x64/Thread.cpp
|
||||
x64/Syscalls.cpp
|
||||
x64/Time.cpp
|
||||
Syscalls/EPoll.cpp
|
||||
Syscalls/FD.cpp
|
||||
Syscalls/FS.cpp
|
||||
Syscalls/Info.cpp
|
||||
Syscalls/IO.cpp
|
||||
Syscalls/Key.cpp
|
||||
Syscalls/Memory.cpp
|
||||
Syscalls/Msg.cpp
|
||||
Syscalls/Sched.cpp
|
||||
Syscalls/Semaphore.cpp
|
||||
Syscalls/SHM.cpp
|
||||
Syscalls/Signals.cpp
|
||||
Syscalls/Socket.cpp
|
||||
Syscalls/Thread.cpp
|
||||
Syscalls/Time.cpp
|
||||
Syscalls/Timer.cpp
|
||||
Syscalls/NotImplemented.cpp
|
||||
Syscalls/Stubs.cpp
|
||||
)
|
||||
|
||||
target_link_libraries(LinuxEmulation FEXCore pthread numa)
|
||||
@@ -13,6 +13,7 @@
|
||||
#include <sys/uio.h>
|
||||
#include <sys/vfs.h>
|
||||
#include <unistd.h>
|
||||
#include <sys/syscall.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
|
||||
@@ -83,15 +84,15 @@ uint64_t FileManager::Access(const char *pathname, [[maybe_unused]] int mode) {
|
||||
return ::access(pathname, mode);
|
||||
}
|
||||
|
||||
uint64_t FileManager::FAccessat(int dirfd, const char *pathname, int mode, int flags) {
|
||||
uint64_t FileManager::FAccessat(int dirfd, const char *pathname, int mode) {
|
||||
auto Path = GetEmulatedPath(pathname);
|
||||
if (!Path.empty()) {
|
||||
uint64_t Result = ::faccessat(dirfd, Path.c_str(), mode, flags);
|
||||
uint64_t Result = ::syscall(SYS_faccessat, dirfd, Path.c_str(), mode);
|
||||
if (Result != -1)
|
||||
return Result;
|
||||
}
|
||||
|
||||
return ::faccessat(dirfd, pathname, mode, flags);
|
||||
return ::syscall(SYS_faccessat, dirfd, pathname, mode);
|
||||
}
|
||||
|
||||
uint64_t FileManager::Readlink(const char *pathname, char *buf, size_t bufsiz) {
|
||||
|
||||
@@ -27,7 +27,7 @@ public:
|
||||
uint64_t Stat(const char *pathname, void *buf);
|
||||
uint64_t Lstat(const char *path, void *buf);
|
||||
uint64_t Access(const char *pathname, int mode);
|
||||
uint64_t FAccessat(int dirfd, const char *pathname, int mode, int flags);
|
||||
uint64_t FAccessat(int dirfd, const char *pathname, int mode);
|
||||
uint64_t Readlink(const char *pathname, char *buf, size_t bufsiz);
|
||||
uint64_t Chmod(const char *pathname, mode_t mode);
|
||||
uint64_t Readlinkat(int dirfd, const char *pathname, char *buf, size_t bufsiz);
|
||||
|
||||
@@ -20,18 +20,13 @@
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define BRK_SIZE 0x1000'0000
|
||||
|
||||
namespace FEX::HLE {
|
||||
SyscallHandler *_SyscallHandler{};
|
||||
|
||||
uint64_t SyscallHandler::HandleBRK(FEXCore::Core::InternalThreadState *Thread, void *Addr) {
|
||||
std::lock_guard<std::mutex> lk(MMapMutex);
|
||||
uint64_t Result;
|
||||
|
||||
if (DataSpace == 0) {
|
||||
DefaultProgramBreak(Thread);
|
||||
}
|
||||
uint64_t Result;
|
||||
|
||||
if (Addr == nullptr) { // Just wants to get the location of the program break atm
|
||||
Result = DataSpace + DataSpaceSize;
|
||||
@@ -46,10 +41,45 @@ uint64_t SyscallHandler::HandleBRK(FEXCore::Core::InternalThreadState *Thread, v
|
||||
}
|
||||
else {
|
||||
uint64_t NewSize = NewEnd - DataSpace;
|
||||
uint64_t NewSizeAligned = AlignUp(NewSize, 4096);
|
||||
|
||||
// make sure we don't overflow to TLS storage
|
||||
if (NewSize >= BRK_SIZE)
|
||||
return -ENOMEM;
|
||||
if (NewSizeAligned < DataSpaceMaxSize) {
|
||||
// If we are shrinking the brk then munmap the ranges
|
||||
// That way we gain the memory back and also give the application zero pages if it allocates again
|
||||
// DataspaceMaxSize is always page aligned
|
||||
|
||||
uint64_t RemainingSize = DataSpaceMaxSize - NewSizeAligned;
|
||||
// We have pages we can unmap
|
||||
munmap(reinterpret_cast<void*>(DataSpace + NewSizeAligned), RemainingSize);
|
||||
DataSpaceMaxSize = NewSizeAligned;
|
||||
}
|
||||
else if (NewSize > DataSpaceMaxSize) {
|
||||
constexpr static uint64_t SizeAlignment = 8 * 1024 * 1024;
|
||||
uint64_t AllocateNewSize = AlignUp(NewSize, SizeAlignment) - DataSpaceMaxSize;
|
||||
if (!Is64BitMode() &&
|
||||
(DataSpace + DataSpaceMaxSize + AllocateNewSize > 0x1'0000'0000ULL)) {
|
||||
// If we are 32bit and we tried going about the 32bit limit then out of memory
|
||||
return DataSpace + DataSpaceSize;
|
||||
}
|
||||
|
||||
uint64_t NewBRK = (uint64_t)mmap((void*)(DataSpace + DataSpaceMaxSize), AllocateNewSize, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
if (NewBRK != (DataSpace + DataSpaceMaxSize)) {
|
||||
// Couldn't allocate that the region we wanted
|
||||
// Can happen if MAP_FIXED_NOREPLACE isn't understood by the kernel
|
||||
munmap(reinterpret_cast<void*>(NewBRK), AllocateNewSize);
|
||||
NewBRK = ~0ULL;
|
||||
}
|
||||
|
||||
if (NewBRK == ~0ULL) {
|
||||
// If we couldn't allocate a new region then out of memory
|
||||
return DataSpace + DataSpaceSize;
|
||||
}
|
||||
else {
|
||||
// Increase our BRK size
|
||||
DataSpaceMaxSize += AllocateNewSize;
|
||||
}
|
||||
}
|
||||
|
||||
DataSpaceSize = NewSize;
|
||||
}
|
||||
@@ -58,9 +88,10 @@ uint64_t SyscallHandler::HandleBRK(FEXCore::Core::InternalThreadState *Thread, v
|
||||
return Result;
|
||||
}
|
||||
|
||||
void SyscallHandler::DefaultProgramBreak(FEXCore::Core::InternalThreadState *Thread) {
|
||||
DataSpaceSize = 0;
|
||||
DataSpace = (uint64_t)mmap(nullptr, BRK_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
void SyscallHandler::DefaultProgramBreak(uint64_t Base, uint64_t Size) {
|
||||
DataSpace = Base;
|
||||
DataSpaceMaxSize = Size;
|
||||
DataSpaceStartingSize = Size;
|
||||
}
|
||||
|
||||
SyscallHandler::SyscallHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *_SignalDelegation)
|
||||
@@ -70,6 +101,10 @@ SyscallHandler::SyscallHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalD
|
||||
HostKernelVersion = CalculateHostKernelVersion();
|
||||
}
|
||||
|
||||
SyscallHandler::~SyscallHandler() {
|
||||
munmap(reinterpret_cast<void*>(DataSpace + DataSpaceStartingSize), DataSpaceMaxSize - DataSpaceStartingSize);
|
||||
}
|
||||
|
||||
uint32_t SyscallHandler::CalculateHostKernelVersion() {
|
||||
struct utsname buf{};
|
||||
if (uname(&buf) == -1) {
|
||||
|
||||
@@ -47,12 +47,12 @@ namespace FEX::HLE {
|
||||
class SyscallHandler : public FEXCore::HLE::SyscallHandler {
|
||||
public:
|
||||
SyscallHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *_SignalDelegation);
|
||||
virtual ~SyscallHandler() = default;
|
||||
virtual ~SyscallHandler();
|
||||
|
||||
// In the case that the syscall doesn't hit the optimized path then we still need to go here
|
||||
uint64_t HandleSyscall(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args) final override;
|
||||
|
||||
void DefaultProgramBreak(FEXCore::Core::InternalThreadState *Thread);
|
||||
void DefaultProgramBreak(uint64_t Base, uint64_t Size);
|
||||
|
||||
using SyscallPtrArg0 = uint64_t(*)(FEXCore::Core::InternalThreadState *Thread);
|
||||
using SyscallPtrArg1 = uint64_t(*)(FEXCore::Core::InternalThreadState *Thread, uint64_t);
|
||||
@@ -116,6 +116,8 @@ protected:
|
||||
// BRK management
|
||||
uint64_t DataSpace {};
|
||||
uint64_t DataSpaceSize {};
|
||||
uint64_t DataSpaceMaxSize {};
|
||||
uint64_t DataSpaceStartingSize{};
|
||||
|
||||
// (Major << 24) | (Minor << 16) | Patch
|
||||
uint32_t HostKernelVersion{};
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#include <sys/eventfd.h>
|
||||
#include <sys/syscall.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
static int RemapFlags(int flags) {
|
||||
@@ -217,13 +218,13 @@ namespace FEX::HLE {
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL(fchmodat, [](FEXCore::Core::InternalThreadState *Thread, int dirfd, const char *pathname, mode_t mode, int flags) -> uint64_t {
|
||||
uint64_t Result = fchmodat(dirfd, pathname, mode, flags);
|
||||
REGISTER_SYSCALL_IMPL(fchmodat, [](FEXCore::Core::InternalThreadState *Thread, int dirfd, const char *pathname, mode_t mode) -> uint64_t {
|
||||
uint64_t Result = syscall(SYS_fchmodat, dirfd, pathname, mode);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL(faccessat, [](FEXCore::Core::InternalThreadState *Thread, int dirfd, const char *pathname, int mode, int flags) -> uint64_t {
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.FAccessat(dirfd, pathname, mode, flags);
|
||||
REGISTER_SYSCALL_IMPL(faccessat, [](FEXCore::Core::InternalThreadState *Thread, int dirfd, const char *pathname, int mode) -> uint64_t {
|
||||
uint64_t Result = FEX::HLE::_SyscallHandler->FM.FAccessat(dirfd, pathname, mode);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#include <signal.h>
|
||||
#include <sys/time.h>
|
||||
#include <time.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
@@ -18,27 +19,27 @@ namespace FEX::HLE {
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL(timer_create, [](FEXCore::Core::InternalThreadState *Thread, clockid_t clockid, struct sigevent *sevp, timer_t *timerid) -> uint64_t {
|
||||
uint64_t Result = ::timer_create(clockid, sevp, timerid);
|
||||
uint64_t Result = ::syscall(SYS_timer_create, clockid, sevp, timerid);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL(timer_settime, [](FEXCore::Core::InternalThreadState *Thread, timer_t timerid, int flags, const struct itimerspec *new_value, struct itimerspec *old_value) -> uint64_t {
|
||||
uint64_t Result = ::timer_settime(timerid, flags, new_value, old_value);
|
||||
uint64_t Result = ::syscall(SYS_timer_settime, timerid, flags, new_value, old_value);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL(timer_gettime, [](FEXCore::Core::InternalThreadState *Thread, timer_t timerid, struct itimerspec *curr_value) -> uint64_t {
|
||||
uint64_t Result = ::timer_gettime(timerid, curr_value);
|
||||
uint64_t Result = ::syscall(SYS_timer_gettime, timerid, curr_value);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL(timer_getoverrun, [](FEXCore::Core::InternalThreadState *Thread, timer_t timerid) -> uint64_t {
|
||||
uint64_t Result = ::timer_getoverrun(timerid);
|
||||
uint64_t Result = ::syscall(SYS_timer_getoverrun, timerid);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL(timer_delete, [](FEXCore::Core::InternalThreadState *Thread, timer_t timerid) -> uint64_t {
|
||||
uint64_t Result = ::timer_delete(timerid);
|
||||
uint64_t Result = ::syscall(SYS_timer_delete, timerid);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
|
||||
@@ -25,6 +25,28 @@ ARG_TO_STR(FEX::HLE::x32::compat_ptr<FEX::HLE::x32::sigset_argpack32>, "%lx")
|
||||
|
||||
namespace FEX::HLE::x32 {
|
||||
using fd_set32 = uint32_t;
|
||||
#ifdef _M_X86_64
|
||||
uint32_t ioctl32(int fd, uint32_t request, uint32_t args) {
|
||||
uint32_t Result{};
|
||||
// x86-64 with compatibility compiled in can still reach the 32bit syscall handler through int 0x80
|
||||
// It can also access through the x32 syscall API which will eventually be deprecated and removed
|
||||
// x32 ioctl number is ((1U << 30) | 514)
|
||||
#define NR_ioctl_x86 54
|
||||
__asm volatile("int $0x80;"
|
||||
: "=a" (Result)
|
||||
: "a" (NR_ioctl_x86)
|
||||
, "b" (fd)
|
||||
, "c" (request)
|
||||
, "d" (args)
|
||||
: "memory");
|
||||
return Result;
|
||||
}
|
||||
#else
|
||||
uint32_t ioctl32(int fd, uint32_t request, uint32_t args) {
|
||||
// Not currently implemented on AArch64
|
||||
return -ENOSYS;
|
||||
}
|
||||
#endif
|
||||
|
||||
void RegisterFD() {
|
||||
REGISTER_SYSCALL_IMPL_X32(poll, [](FEXCore::Core::InternalThreadState *Thread, struct pollfd *fds, nfds_t nfds, int timeout) -> uint64_t {
|
||||
@@ -359,11 +381,7 @@ namespace FEX::HLE::x32 {
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(ioctl, [](FEXCore::Core::InternalThreadState *Thread, int fd, uint32_t request, uint32_t args) -> uint64_t {
|
||||
uint64_t Result = ::syscall(SYS_ioctl,
|
||||
static_cast<uint64_t>(fd),
|
||||
request,
|
||||
args);
|
||||
SYSCALL_ERRNO();
|
||||
return ioctl32(fd, request, args);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(getdents, [](FEXCore::Core::InternalThreadState *Thread, int fd, void *dirp, uint32_t count) -> uint64_t {
|
||||
|
||||
@@ -1,277 +1,29 @@
|
||||
#include "Tests/LinuxSyscalls/Syscalls.h"
|
||||
#include "Tests/LinuxSyscalls/x32/Syscalls.h"
|
||||
|
||||
#include "Common/MathUtils.h"
|
||||
|
||||
#include <bitset>
|
||||
#include <map>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/ipc.h>
|
||||
|
||||
namespace FEX::HLE::x32 {
|
||||
class MemAllocator {
|
||||
private:
|
||||
static constexpr uint64_t PAGE_SHIFT = 12;
|
||||
static constexpr uint64_t PAGE_SIZE = 1 << PAGE_SHIFT;
|
||||
static constexpr uint64_t PAGE_MASK = (1 << PAGE_SHIFT) - 1;
|
||||
static constexpr uint64_t BASE_KEY = 16;
|
||||
const uint64_t TOP_KEY = 0xFFFF'F000ULL >> PAGE_SHIFT;
|
||||
|
||||
public:
|
||||
MemAllocator() {
|
||||
// First 16 pages are taken by the Linux kernel
|
||||
for (size_t i = 0; i < 16; ++i) {
|
||||
MappedPages.set(i);
|
||||
}
|
||||
// Take the top page as well
|
||||
MappedPages.set(TOP_KEY);
|
||||
}
|
||||
void *mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset);
|
||||
int munmap(void *addr, size_t length);
|
||||
void *mremap(void *old_address, size_t old_size, size_t new_size, int flags, void *new_address);
|
||||
|
||||
private:
|
||||
// Set that contains 4k mapped pages
|
||||
// This is the full 32bit memory range
|
||||
std::bitset<0x10'0000> MappedPages;
|
||||
uint64_t LastScanLocation = BASE_KEY;
|
||||
std::mutex AllocMutex{};
|
||||
uint64_t FindPageRange(uint64_t Start, size_t Pages);
|
||||
uint64_t FindPageRange_TopDown(uint64_t Start, size_t Pages);
|
||||
};
|
||||
|
||||
uint64_t MemAllocator::FindPageRange(uint64_t Start, size_t Pages) {
|
||||
// Linear range scan
|
||||
while (Start != TOP_KEY) {
|
||||
bool Free = true;
|
||||
if ((Start + Pages) > TOP_KEY) {
|
||||
return 0;
|
||||
}
|
||||
uint64_t Offset = 0;
|
||||
for (; Offset < Pages; ++Offset) {
|
||||
if (MappedPages.test(Start + Offset)) {
|
||||
Free = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Free) {
|
||||
return Start;
|
||||
}
|
||||
Start += Offset + 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t MemAllocator::FindPageRange_TopDown(uint64_t Start, size_t Pages) {
|
||||
// Linear range scan
|
||||
Start -= Pages;
|
||||
while (Start != BASE_KEY) {
|
||||
bool Free = true;
|
||||
if (Start < BASE_KEY) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t Offset = 0;
|
||||
for (; Offset < Pages; ++Offset) {
|
||||
if (MappedPages.test(Start + Offset)) {
|
||||
Free = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Free) {
|
||||
return Start;
|
||||
}
|
||||
Start -= (Pages - Offset) + 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
void *MemAllocator::mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
std::scoped_lock<std::mutex> lk{AllocMutex};
|
||||
size_t PagesLength = AlignUp(length, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
|
||||
uintptr_t Addr = reinterpret_cast<uintptr_t>(addr);
|
||||
uintptr_t PageAddr = AlignUp(Addr, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
|
||||
uintptr_t PageEnd = PageAddr + PagesLength;
|
||||
|
||||
bool Fixed = ((flags & MAP_FIXED) ||
|
||||
(flags & MAP_FIXED_NOREPLACE));
|
||||
|
||||
// Both Addr and length must be page aligned
|
||||
if (Addr & PAGE_MASK) {
|
||||
return (void*)-EINVAL;
|
||||
}
|
||||
|
||||
// If we do have an fd then offset must be page aligned
|
||||
if (fd != -1 &&
|
||||
offset & PAGE_MASK) {
|
||||
return (void*)-EINVAL;
|
||||
}
|
||||
|
||||
if (Addr + length > std::numeric_limits<uint32_t>::max()) {
|
||||
return (void*)-EOVERFLOW;
|
||||
}
|
||||
|
||||
// Check reserved range
|
||||
if (Fixed && PageAddr < 16) {
|
||||
return (void*)-EINVAL;
|
||||
}
|
||||
|
||||
if (!Fixed) {
|
||||
// If we aren't mapping fixed the ignore the address input
|
||||
Addr = 0;
|
||||
PageAddr = 0;
|
||||
PageEnd = PagesLength;
|
||||
}
|
||||
|
||||
// Find a region that fits our address
|
||||
if (Addr == 0) {
|
||||
bool Wrapped = false;
|
||||
uint64_t BottomPage = LastScanLocation;
|
||||
restart:
|
||||
{
|
||||
// Linear range scan
|
||||
uint64_t LowerPage = FindPageRange(BottomPage, PagesLength);
|
||||
if (LowerPage == 0) {
|
||||
// Try again but this time from the start
|
||||
BottomPage = BASE_KEY;
|
||||
LowerPage = FindPageRange(BottomPage, PagesLength);
|
||||
}
|
||||
|
||||
uint64_t UpperPage = LowerPage + PagesLength;
|
||||
if (LowerPage == 0) {
|
||||
return (void*)(uintptr_t)-ENOMEM;
|
||||
}
|
||||
{
|
||||
// Try and map the range
|
||||
void *MappedPtr = ::mmap(
|
||||
reinterpret_cast<void*>(LowerPage<< PAGE_SHIFT),
|
||||
length,
|
||||
prot,
|
||||
flags | MAP_FIXED_NOREPLACE,
|
||||
fd,
|
||||
offset);
|
||||
|
||||
if (MappedPtr == MAP_FAILED) {
|
||||
if (UpperPage == TOP_KEY) {
|
||||
BottomPage = BASE_KEY;
|
||||
Wrapped = true;
|
||||
goto restart;
|
||||
}
|
||||
else if (Wrapped &&
|
||||
LowerPage >= LastScanLocation) {
|
||||
// We linear scanned the entire memory range. Give up
|
||||
return (void*)(uintptr_t)-errno;
|
||||
}
|
||||
else {
|
||||
// Try again
|
||||
BottomPage += PagesLength;
|
||||
goto restart;
|
||||
}
|
||||
}
|
||||
else {
|
||||
LastScanLocation = UpperPage;
|
||||
// Set the range as mapped
|
||||
for (size_t i = 0; i < PagesLength; ++i) {
|
||||
MappedPages.set(LowerPage + i);
|
||||
}
|
||||
return MappedPtr;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
void *MappedPtr = ::mmap(
|
||||
reinterpret_cast<void*>(PageAddr << PAGE_SHIFT),
|
||||
PagesLength << PAGE_SHIFT,
|
||||
prot,
|
||||
flags,
|
||||
fd,
|
||||
offset);
|
||||
|
||||
if (MappedPtr != MAP_FAILED) {
|
||||
for (size_t i = 0; i < PagesLength; ++i) {
|
||||
MappedPages.set(PageAddr + i);
|
||||
}
|
||||
return MappedPtr;
|
||||
}
|
||||
else {
|
||||
return (void*)(uintptr_t)-errno;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int MemAllocator::munmap(void *addr, size_t length) {
|
||||
std::scoped_lock<std::mutex> lk{AllocMutex};
|
||||
size_t PagesLength = AlignUp(length, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
|
||||
uintptr_t Addr = reinterpret_cast<uintptr_t>(addr);
|
||||
uintptr_t PageAddr = Addr >> PAGE_SHIFT;
|
||||
|
||||
uintptr_t PageEnd = PageAddr + PagesLength;
|
||||
|
||||
// Both Addr and length must be page aligned
|
||||
if (Addr & PAGE_MASK) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if (length & PAGE_MASK) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if (Addr + length > std::numeric_limits<uint32_t>::max()) {
|
||||
return -EOVERFLOW;
|
||||
}
|
||||
|
||||
// Check reserved range
|
||||
if (PageAddr < 16) {
|
||||
// Return success for these
|
||||
return 0;
|
||||
}
|
||||
|
||||
while (PageAddr != PageEnd) {
|
||||
// Always pass to munmap, it may be something allocated we aren't tracking
|
||||
int Result = ::munmap(reinterpret_cast<void*>(PageAddr << PAGE_SHIFT), PAGE_SIZE);
|
||||
if (Result != 0) {
|
||||
return -errno;
|
||||
}
|
||||
|
||||
if (MappedPages.test(PageAddr)) {
|
||||
MappedPages.reset(PageAddr);
|
||||
}
|
||||
|
||||
++PageAddr;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
void *MemAllocator::mremap(void *old_address, size_t old_size, size_t new_size, int flags, void *new_address) {
|
||||
std::scoped_lock<std::mutex> lk{AllocMutex};
|
||||
// XXX: Not currently supported
|
||||
return reinterpret_cast<void*>(-ENOMEM);
|
||||
}
|
||||
|
||||
static std::unique_ptr<MemAllocator> alloc{};
|
||||
void RegisterMemory() {
|
||||
alloc = std::make_unique<MemAllocator>();
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(mmap, [](FEXCore::Core::InternalThreadState *Thread, uint32_t addr, uint32_t length, int prot, int flags, int fd, int32_t offset) -> uint64_t {
|
||||
return (uint64_t)alloc->mmap(reinterpret_cast<void*>(addr), length, prot,flags, fd, offset);
|
||||
return (uint64_t)static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
mmap(reinterpret_cast<void*>(addr), length, prot,flags, fd, offset);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(mmap2, [](FEXCore::Core::InternalThreadState *Thread, uint32_t addr, uint32_t length, int prot, int flags, int fd, int32_t pgoffset) -> uint64_t {
|
||||
return (uint64_t)alloc->mmap(reinterpret_cast<void*>(addr), length, prot,flags, fd, pgoffset * 0x1000);
|
||||
REGISTER_SYSCALL_IMPL_X32(mmap2, [](FEXCore::Core::InternalThreadState *Thread, uint32_t addr, uint32_t length, int prot, int flags, int fd, uint32_t pgoffset) -> uint64_t {
|
||||
return (uint64_t)static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
mmap(reinterpret_cast<void*>(addr), length, prot,flags, fd, (uint64_t)pgoffset * 0x1000);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(munmap, [](FEXCore::Core::InternalThreadState *Thread, void *addr, size_t length) -> uint64_t {
|
||||
return alloc->munmap(addr, length);
|
||||
return static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
munmap(addr, length);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(mprotect, [](FEXCore::Core::InternalThreadState *Thread, void *addr, uint32_t len, int prot) -> uint64_t {
|
||||
@@ -280,7 +32,8 @@ void *MemAllocator::mremap(void *old_address, size_t old_size, size_t new_size,
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(mremap, [](FEXCore::Core::InternalThreadState *Thread, void *old_address, size_t old_size, size_t new_size, int flags, void *new_address) -> uint64_t {
|
||||
return reinterpret_cast<uint64_t>(alloc->mremap(old_address, old_size, new_size, flags, new_address));
|
||||
return reinterpret_cast<uint64_t>(static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
mremap(old_address, old_size, new_size, flags, new_address));
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(mlockall, [](FEXCore::Core::InternalThreadState *Thread, int flags) -> uint64_t {
|
||||
|
||||
@@ -688,23 +688,13 @@ namespace FEX::HLE::x32 {
|
||||
break;
|
||||
}
|
||||
case OP_SHMAT: {
|
||||
Result = reinterpret_cast<uint64_t>(shmat(first, reinterpret_cast<const void*>(ptr), second));
|
||||
if (Result != -1) {
|
||||
uint32_t SmallRet = Result >> 32;
|
||||
if (!(SmallRet == 0 ||
|
||||
SmallRet == ~0U)) {
|
||||
LogMan::Msg::A("Syscall returning something with data in the upper 32bits! BUG!");
|
||||
return -ENOMEM;
|
||||
}
|
||||
|
||||
*reinterpret_cast<uint32_t*>(third) = static_cast<uint32_t>(Result);
|
||||
// Zero return on success
|
||||
Result = 0;
|
||||
}
|
||||
Result = static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
shmat(first, reinterpret_cast<const void*>(ptr), second, reinterpret_cast<uint32_t*>(third));
|
||||
break;
|
||||
}
|
||||
case OP_SHMDT: {
|
||||
Result = ::shmdt(reinterpret_cast<void*>(ptr));
|
||||
Result = static_cast<FEX::HLE::x32::x32SyscallHandler*>(FEX::HLE::_SyscallHandler)->GetAllocator()->
|
||||
shmdt(reinterpret_cast<void*>(ptr));
|
||||
break;
|
||||
}
|
||||
case OP_SHMGET: {
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
#include "Common/MathUtils.h"
|
||||
#include "Tests/LinuxSyscalls/Syscalls.h"
|
||||
#include "Tests/LinuxSyscalls/x32/Syscalls.h"
|
||||
|
||||
@@ -6,9 +7,468 @@
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <map>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/shm.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
#ifndef MREMAP_DONTUNMAP
|
||||
#define MREMAP_DONTUNMAP 4
|
||||
#endif
|
||||
|
||||
namespace FEX::HLE::x32 {
|
||||
uint64_t MemAllocator::FindPageRange(uint64_t Start, size_t Pages) {
|
||||
// Linear range scan
|
||||
while (Start != TOP_KEY) {
|
||||
bool Free = true;
|
||||
if ((Start + Pages) > TOP_KEY) {
|
||||
return 0;
|
||||
}
|
||||
uint64_t Offset = 0;
|
||||
for (; Offset < Pages; ++Offset) {
|
||||
if (MappedPages.test(Start + Offset)) {
|
||||
Free = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Free) {
|
||||
return Start;
|
||||
}
|
||||
Start += Offset + 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t MemAllocator::FindPageRange_TopDown(uint64_t Start, size_t Pages) {
|
||||
// Linear range scan
|
||||
while (Start >= BASE_KEY &&
|
||||
Start <= TOP_KEY) {
|
||||
bool Free = true;
|
||||
|
||||
uint64_t Offset = 0;
|
||||
for (; Offset < Pages; ++Offset) {
|
||||
if (MappedPages.test(Start - Offset)) {
|
||||
Free = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Free) {
|
||||
return Start - Offset;
|
||||
}
|
||||
Start -= Offset + 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
void *MemAllocator::mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
std::scoped_lock<std::mutex> lk{AllocMutex};
|
||||
size_t PagesLength = AlignUp(length, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
|
||||
uintptr_t Addr = reinterpret_cast<uintptr_t>(addr);
|
||||
uintptr_t PageAddr = Addr >> PAGE_SHIFT;
|
||||
|
||||
uintptr_t PageEnd = PageAddr + PagesLength;
|
||||
|
||||
bool Fixed = ((flags & MAP_FIXED) ||
|
||||
(flags & MAP_FIXED_NOREPLACE));
|
||||
|
||||
// Both Addr and length must be page aligned
|
||||
if (Addr & PAGE_MASK) {
|
||||
return reinterpret_cast<void*>(-EINVAL);
|
||||
}
|
||||
|
||||
// If we do have an fd then offset must be page aligned
|
||||
if (fd != -1 &&
|
||||
offset & PAGE_MASK) {
|
||||
return reinterpret_cast<void*>(-EINVAL);
|
||||
}
|
||||
|
||||
if (Addr + length > std::numeric_limits<uint32_t>::max()) {
|
||||
return reinterpret_cast<void*>(-EOVERFLOW);
|
||||
}
|
||||
|
||||
// Check reserved range
|
||||
if (Fixed && PageAddr < 16) {
|
||||
return reinterpret_cast<void*>(-EINVAL);
|
||||
}
|
||||
|
||||
if (!Fixed) {
|
||||
// If we aren't mapping fixed the ignore the address input
|
||||
Addr = 0;
|
||||
PageAddr = 0;
|
||||
PageEnd = PagesLength;
|
||||
}
|
||||
|
||||
// Find a region that fits our address
|
||||
if (Addr == 0) {
|
||||
bool Wrapped = false;
|
||||
uint64_t BottomPage = LastScanLocation;
|
||||
restart:
|
||||
{
|
||||
// Linear range scan
|
||||
uint64_t LowerPage = (this->*FindPageRangePtr)(BottomPage, PagesLength);
|
||||
if (LowerPage == 0) {
|
||||
// Try again but this time from the start
|
||||
BottomPage = LastKeyLocation;
|
||||
LowerPage = (this->*FindPageRangePtr)(BottomPage, PagesLength);
|
||||
}
|
||||
|
||||
uint64_t UpperPage = LowerPage + PagesLength;
|
||||
if (LowerPage == 0) {
|
||||
return reinterpret_cast<void*>(-ENOMEM);
|
||||
}
|
||||
{
|
||||
// Try and map the range
|
||||
void *MappedPtr = ::mmap(
|
||||
reinterpret_cast<void*>(LowerPage<< PAGE_SHIFT),
|
||||
length,
|
||||
prot,
|
||||
flags | MAP_FIXED_NOREPLACE,
|
||||
fd,
|
||||
offset);
|
||||
|
||||
if (MappedPtr == MAP_FAILED &&
|
||||
errno != EEXIST) {
|
||||
return reinterpret_cast<void*>(-errno);
|
||||
}
|
||||
else if (MappedPtr == MAP_FAILED) {
|
||||
if (UpperPage == TOP_KEY) {
|
||||
BottomPage = BASE_KEY;
|
||||
Wrapped = true;
|
||||
goto restart;
|
||||
}
|
||||
else if (Wrapped &&
|
||||
LowerPage >= LastScanLocation) {
|
||||
// We linear scanned the entire memory range. Give up
|
||||
return (void*)(uintptr_t)-errno;
|
||||
}
|
||||
else {
|
||||
// Try again
|
||||
if (SearchDown) {
|
||||
BottomPage -= PagesLength;
|
||||
}
|
||||
else {
|
||||
BottomPage += PagesLength;
|
||||
}
|
||||
goto restart;
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (SearchDown) {
|
||||
LastScanLocation = LowerPage;
|
||||
}
|
||||
else {
|
||||
LastScanLocation = UpperPage;
|
||||
}
|
||||
SetUsedPages(LowerPage, PagesLength);
|
||||
return MappedPtr;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
void *MappedPtr = ::mmap(
|
||||
reinterpret_cast<void*>(PageAddr << PAGE_SHIFT),
|
||||
PagesLength << PAGE_SHIFT,
|
||||
prot,
|
||||
flags,
|
||||
fd,
|
||||
offset);
|
||||
|
||||
if (MappedPtr != MAP_FAILED) {
|
||||
SetUsedPages(PageAddr, PagesLength);
|
||||
return MappedPtr;
|
||||
}
|
||||
else {
|
||||
return reinterpret_cast<void*>(-errno);
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int MemAllocator::munmap(void *addr, size_t length) {
|
||||
std::scoped_lock<std::mutex> lk{AllocMutex};
|
||||
size_t PagesLength = AlignUp(length, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
|
||||
uintptr_t Addr = reinterpret_cast<uintptr_t>(addr);
|
||||
uintptr_t PageAddr = Addr >> PAGE_SHIFT;
|
||||
|
||||
uintptr_t PageEnd = PageAddr + PagesLength;
|
||||
|
||||
// Both Addr and length must be page aligned
|
||||
if (Addr & PAGE_MASK) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if (length & PAGE_MASK) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if (Addr + length > std::numeric_limits<uint32_t>::max()) {
|
||||
return -EOVERFLOW;
|
||||
}
|
||||
|
||||
// Check reserved range
|
||||
if (PageAddr < 16) {
|
||||
// Return success for these
|
||||
return 0;
|
||||
}
|
||||
|
||||
while (PageAddr != PageEnd) {
|
||||
// Always pass to munmap, it may be something allocated we aren't tracking
|
||||
int Result = ::munmap(reinterpret_cast<void*>(PageAddr << PAGE_SHIFT), PAGE_SIZE);
|
||||
if (Result != 0) {
|
||||
return -errno;
|
||||
}
|
||||
|
||||
if (MappedPages.test(PageAddr)) {
|
||||
MappedPages.reset(PageAddr);
|
||||
}
|
||||
|
||||
++PageAddr;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
void *MemAllocator::mremap(void *old_address, size_t old_size, size_t new_size, int flags, void *new_address) {
|
||||
size_t OldPagesLength = AlignUp(old_size, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
size_t NewPagesLength = AlignUp(new_size, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
|
||||
{
|
||||
std::scoped_lock<std::mutex> lk{AllocMutex};
|
||||
if (flags & MREMAP_FIXED) {
|
||||
void *MappedPtr = ::mremap(old_address, old_size, new_size, flags, new_address);
|
||||
|
||||
if (MappedPtr != MAP_FAILED) {
|
||||
if (!(flags & MREMAP_DONTUNMAP)) {
|
||||
// Unmap the old location
|
||||
uintptr_t OldAddr = reinterpret_cast<uintptr_t>(old_address);
|
||||
SetFreePages(OldAddr >> PAGE_SHIFT, OldPagesLength);
|
||||
}
|
||||
|
||||
// Map the new pages
|
||||
uintptr_t NewAddr = reinterpret_cast<uintptr_t>(MappedPtr);
|
||||
SetUsedPages(NewAddr >> PAGE_SHIFT, NewPagesLength);
|
||||
}
|
||||
else {
|
||||
return reinterpret_cast<void*>(-errno);
|
||||
}
|
||||
}
|
||||
else {
|
||||
uintptr_t OldAddr = reinterpret_cast<uintptr_t>(old_address);
|
||||
uintptr_t OldPageAddr = OldAddr >> PAGE_SHIFT;
|
||||
|
||||
if (NewPagesLength < OldPagesLength) {
|
||||
void *MappedPtr = ::mremap(old_address, old_size, new_size, flags & ~MREMAP_MAYMOVE);
|
||||
|
||||
if (MappedPtr != MAP_FAILED) {
|
||||
// Clear the pages that we just shrunk
|
||||
size_t NewPagesLength = AlignUp(new_size, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
uintptr_t NewPageAddr = reinterpret_cast<uintptr_t>(MappedPtr) >> PAGE_SHIFT;
|
||||
SetFreePages(NewPageAddr + NewPagesLength, OldPagesLength - NewPagesLength);
|
||||
return MappedPtr;
|
||||
}
|
||||
else {
|
||||
return reinterpret_cast<void*>(-errno);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Scan the region forward from our first region's endd to see if it can be extended
|
||||
bool CanExtend{true};
|
||||
|
||||
for (size_t i = OldPagesLength; i < NewPagesLength; ++i) {
|
||||
if (MappedPages[OldPageAddr + i]) {
|
||||
CanExtend = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (CanExtend) {
|
||||
void *MappedPtr = ::mremap(old_address, old_size, new_size, flags & ~MREMAP_MAYMOVE);
|
||||
|
||||
if (MappedPtr != MAP_FAILED) {
|
||||
// Map the new pages
|
||||
size_t NewPagesLength = AlignUp(new_size, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
uintptr_t NewAddr = reinterpret_cast<uintptr_t>(MappedPtr);
|
||||
SetUsedPages(NewAddr >> PAGE_SHIFT, NewPagesLength);
|
||||
return MappedPtr;
|
||||
}
|
||||
else if (!(flags & MREMAP_MAYMOVE)) {
|
||||
// We have one more chance if MAYMOVE is specified
|
||||
return reinterpret_cast<void*>(-errno);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Flags can not contain MREMAP_FIXED at this point
|
||||
// Flags might contain MREMAP_MAYMOVE and/or MREMAP_DONTUNMAP
|
||||
// New Size is >= old size
|
||||
|
||||
// First, try and allocate a region the size of the new size
|
||||
void *MappedPtr = this->mmap(nullptr, new_size, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
std::scoped_lock<std::mutex> lk{AllocMutex};
|
||||
if (reinterpret_cast<uintptr_t>(MappedPtr) > -4096) {
|
||||
// Couldn't find a region that fit our space
|
||||
return MappedPtr;
|
||||
}
|
||||
|
||||
// Good news, we found a region
|
||||
// This will overwrite the previous mmap if it succeeds
|
||||
MappedPtr = ::mremap(old_address, old_size, new_size, flags | MREMAP_FIXED | MREMAP_MAYMOVE, MappedPtr);
|
||||
|
||||
if (MappedPtr != MAP_FAILED) {
|
||||
if (!(flags & MREMAP_DONTUNMAP) &&
|
||||
MappedPtr != old_address) {
|
||||
// If we have both MREMAP_DONTUNMAP not set and the new pointer is at a new location
|
||||
// Make sure to clear the old mapping
|
||||
uintptr_t OldAddr = reinterpret_cast<uintptr_t>(old_address);
|
||||
SetFreePages(OldAddr >> PAGE_SHIFT , OldPagesLength);
|
||||
}
|
||||
|
||||
// Map the new pages
|
||||
size_t NewPagesLength = AlignUp(new_size, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
uintptr_t NewAddr = reinterpret_cast<uintptr_t>(MappedPtr);
|
||||
SetUsedPages(NewAddr >> PAGE_SHIFT, NewPagesLength);
|
||||
return MappedPtr;
|
||||
}
|
||||
|
||||
// Failed
|
||||
return reinterpret_cast<void*>(-errno);
|
||||
}
|
||||
|
||||
uint64_t MemAllocator::shmat(int shmid, const void* shmaddr, int shmflg, uint32_t *ResultAddress) {
|
||||
std::scoped_lock<std::mutex> lk{AllocMutex};
|
||||
|
||||
if (shmaddr != nullptr) {
|
||||
// shmaddr must be valid
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(::shmat(shmid, shmaddr, shmflg));
|
||||
if (Result != -1) {
|
||||
uint32_t SmallRet = Result >> 32;
|
||||
if (!(SmallRet == 0 ||
|
||||
SmallRet == ~0U)) {
|
||||
LogMan::Msg::A("Syscall returning something with data in the upper 32bits! BUG!");
|
||||
return -ENOMEM;
|
||||
}
|
||||
|
||||
uintptr_t NewAddr = reinterpret_cast<uintptr_t>(Result);
|
||||
uintptr_t NewPageAddr = NewAddr >> PAGE_SHIFT;
|
||||
|
||||
// Add to the map
|
||||
PageToShm[NewPageAddr] = shmid;
|
||||
|
||||
*ResultAddress = Result;
|
||||
|
||||
// We must get the shm size and track it
|
||||
struct shmid_ds buf{};
|
||||
|
||||
if (shmctl(shmid, IPC_STAT, &buf) == 0) {
|
||||
// Map the new pages
|
||||
size_t NewPagesLength = buf.shm_segsz >> PAGE_SHIFT;
|
||||
SetUsedPages(NewPageAddr, NewPagesLength);
|
||||
}
|
||||
|
||||
// Zero on working result
|
||||
Result = 0;
|
||||
}
|
||||
else {
|
||||
Result = -errno;
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
else {
|
||||
// We must get the shm size and track it
|
||||
struct shmid_ds buf{};
|
||||
uint64_t PagesLength{};
|
||||
|
||||
if (shmctl(shmid, IPC_STAT, &buf) == 0) {
|
||||
PagesLength = AlignUp(buf.shm_segsz, PAGE_SIZE) >> PAGE_SHIFT;
|
||||
}
|
||||
else {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
bool Wrapped = false;
|
||||
uint64_t BottomPage = LastScanLocation;
|
||||
restart:
|
||||
{
|
||||
// Linear range scan
|
||||
uint64_t LowerPage = (this->*FindPageRangePtr)(BottomPage, PagesLength);
|
||||
if (LowerPage == 0) {
|
||||
// Try again but this time from the start
|
||||
BottomPage = LastKeyLocation;
|
||||
LowerPage = (this->*FindPageRangePtr)(BottomPage, PagesLength);
|
||||
}
|
||||
|
||||
uint64_t UpperPage = LowerPage + PagesLength;
|
||||
if (LowerPage == 0) {
|
||||
return -ENOMEM;
|
||||
}
|
||||
{
|
||||
// Try and map the range
|
||||
void *MappedPtr = ::shmat(
|
||||
shmid,
|
||||
reinterpret_cast<const void*>(LowerPage << PAGE_SHIFT),
|
||||
shmflg);
|
||||
|
||||
if (MappedPtr == MAP_FAILED) {
|
||||
if (UpperPage == TOP_KEY) {
|
||||
BottomPage = LastKeyLocation;
|
||||
Wrapped = true;
|
||||
goto restart;
|
||||
}
|
||||
else if (Wrapped &&
|
||||
LowerPage >= LastScanLocation) {
|
||||
// We linear scanned the entire memory range. Give up
|
||||
return -errno;
|
||||
}
|
||||
else {
|
||||
// Try again
|
||||
BottomPage += PagesLength;
|
||||
goto restart;
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (SearchDown) {
|
||||
LastScanLocation = LowerPage;
|
||||
}
|
||||
else {
|
||||
LastScanLocation = UpperPage;
|
||||
}
|
||||
// Set the range as mapped
|
||||
SetUsedPages(LowerPage, PagesLength);
|
||||
|
||||
*ResultAddress = reinterpret_cast<uint64_t>(MappedPtr);
|
||||
|
||||
// Add to the map
|
||||
PageToShm[LowerPage] = shmid;
|
||||
|
||||
// Zero on working result
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
uint64_t MemAllocator::shmdt(const void* shmaddr) {
|
||||
uint32_t AddrPage = reinterpret_cast<uint64_t>(shmaddr) >> PAGE_SHIFT;
|
||||
auto it = PageToShm.find(AddrPage);
|
||||
|
||||
if (it == PageToShm.end()) {
|
||||
// Page wasn't mapped
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
uint64_t Result = ::shmdt(shmaddr);
|
||||
PageToShm.erase(it);
|
||||
return Result;
|
||||
}
|
||||
|
||||
void RegisterEpoll();
|
||||
void RegisterFD();
|
||||
void RegisterFS();
|
||||
@@ -61,14 +521,6 @@ namespace FEX::HLE::x32 {
|
||||
});
|
||||
}
|
||||
|
||||
class x32SyscallHandler final : public FEX::HLE::SyscallHandler {
|
||||
public:
|
||||
x32SyscallHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *_SignalDelegation);
|
||||
|
||||
private:
|
||||
void RegisterSyscallHandlers();
|
||||
};
|
||||
|
||||
uint32_t Unimplemented(FEXCore::Core::InternalThreadState *Thread, uint64_t SyscallNumber) {
|
||||
auto name = GetSyscallName(SyscallNumber);
|
||||
ERROR_AND_DIE("Unhandled system call: %d, %s", SyscallNumber, name);
|
||||
@@ -77,6 +529,7 @@ private:
|
||||
|
||||
x32SyscallHandler::x32SyscallHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *_SignalDelegation)
|
||||
: SyscallHandler {ctx, _SignalDelegation} {
|
||||
AllocHandler = std::make_unique<MemAllocator>();
|
||||
OSABI = FEXCore::HLE::SyscallOSABI::OS_LINUX32;
|
||||
RegisterSyscallHandlers();
|
||||
}
|
||||
|
||||
@@ -1,17 +1,19 @@
|
||||
#pragma once
|
||||
|
||||
#include "Tests/LinuxSyscalls/Syscalls.h"
|
||||
#include "Tests/LinuxSyscalls/FileManagement.h"
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include "Tests/LinuxSyscalls/x32/Types.h"
|
||||
|
||||
#include <atomic>
|
||||
#include <bitset>
|
||||
#include <condition_variable>
|
||||
#include <map>
|
||||
#include <mutex>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace FEX::HLE {
|
||||
class SignalDelegator;
|
||||
class SyscallHandler;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
@@ -21,6 +23,83 @@ struct InternalThreadState;
|
||||
namespace FEX::HLE::x32 {
|
||||
#include "SyscallsEnum.h"
|
||||
|
||||
class MemAllocator final {
|
||||
private:
|
||||
static constexpr uint64_t PAGE_SHIFT = 12;
|
||||
static constexpr uint64_t PAGE_SIZE = 1 << PAGE_SHIFT;
|
||||
static constexpr uint64_t PAGE_MASK = (1 << PAGE_SHIFT) - 1;
|
||||
static constexpr uint64_t BASE_KEY = 16;
|
||||
const uint64_t TOP_KEY = 0xFFFF'F000ULL >> PAGE_SHIFT;
|
||||
|
||||
public:
|
||||
MemAllocator() {
|
||||
// First 16 pages are taken by the Linux kernel
|
||||
for (size_t i = 0; i < 16; ++i) {
|
||||
MappedPages.set(i);
|
||||
}
|
||||
// Take the top page as well
|
||||
MappedPages.set(TOP_KEY);
|
||||
if (SearchDown) {
|
||||
LastScanLocation = TOP_KEY;
|
||||
LastKeyLocation = TOP_KEY;
|
||||
FindPageRangePtr = &MemAllocator::FindPageRange_TopDown;
|
||||
}
|
||||
else {
|
||||
LastScanLocation = BASE_KEY;
|
||||
LastKeyLocation = BASE_KEY;
|
||||
FindPageRangePtr = &MemAllocator::FindPageRange;
|
||||
}
|
||||
}
|
||||
void *mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset);
|
||||
int munmap(void *addr, size_t length);
|
||||
void *mremap(void *old_address, size_t old_size, size_t new_size, int flags, void *new_address);
|
||||
uint64_t shmat(int shmid, const void* shmaddr, int shmflg, uint32_t *ResultAddress);
|
||||
uint64_t shmdt(const void* shmaddr);
|
||||
static constexpr bool SearchDown = true;
|
||||
|
||||
// PageAddr is a page already shifted to page index
|
||||
// PagesLength is the number of pages
|
||||
void SetUsedPages(uint64_t PageAddr, size_t PagesLength) {
|
||||
// Set the range as mapped
|
||||
for (size_t i = 0; i < PagesLength; ++i) {
|
||||
MappedPages.set(PageAddr + i);
|
||||
}
|
||||
}
|
||||
|
||||
// PageAddr is a page already shifted to page index
|
||||
// PagesLength is the number of pages
|
||||
void SetFreePages(uint64_t PageAddr, size_t PagesLength) {
|
||||
// Set the range as unused
|
||||
for (size_t i = 0; i < PagesLength; ++i) {
|
||||
MappedPages.reset(PageAddr + i);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Set that contains 4k mapped pages
|
||||
// This is the full 32bit memory range
|
||||
std::bitset<0x10'0000> MappedPages;
|
||||
std::map<uint32_t, int> PageToShm{};
|
||||
uint64_t LastScanLocation{};
|
||||
uint64_t LastKeyLocation{};
|
||||
std::mutex AllocMutex{};
|
||||
uint64_t FindPageRange(uint64_t Start, size_t Pages);
|
||||
uint64_t FindPageRange_TopDown(uint64_t Start, size_t Pages);
|
||||
using FindHandler = uint64_t(MemAllocator::*)(uint64_t Start, size_t Pages);
|
||||
FindHandler FindPageRangePtr{};
|
||||
};
|
||||
|
||||
class x32SyscallHandler final : public FEX::HLE::SyscallHandler {
|
||||
public:
|
||||
x32SyscallHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *_SignalDelegation);
|
||||
|
||||
FEX::HLE::x32::MemAllocator *GetAllocator() { return AllocHandler.get(); }
|
||||
|
||||
private:
|
||||
void RegisterSyscallHandlers();
|
||||
std::unique_ptr<MemAllocator> AllocHandler{};
|
||||
};
|
||||
|
||||
FEX::HLE::SyscallHandler *CreateHandler(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *_SignalDelegation);
|
||||
void RegisterSyscallInternal(int SyscallNumber,
|
||||
#ifdef DEBUG_STRACE
|
||||
|
||||
@@ -94,6 +94,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Config::SetConfig(CTX, FEXCore::Config::CONFIG_ABI_LOCAL_FLAGS, ABILocalFlags());
|
||||
FEXCore::Config::SetConfig(CTX, FEXCore::Config::CONFIG_ABI_NO_PF, AbiNoPF());
|
||||
FEXCore::Config::SetConfig(CTX, FEXCore::Config::CONFIG_DUMPIR, DumpIR());
|
||||
FEXCore::Config::SetConfig(CTX, FEXCore::Config::CONFIG_VALIDATE_IR_PARSER, true);
|
||||
FEXCore::Context::SetCustomCPUBackendFactory(CTX, HostFactory::CPUCreationFactory);
|
||||
|
||||
FEXCore::Context::InitializeContext(CTX);
|
||||
|
||||
@@ -57,9 +57,9 @@ namespace {
|
||||
ConfigFilename = {};
|
||||
LoadedConfig = std::make_unique<FEX::Config::EmptyMapper>();
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_DEFAULTCORE, "1");
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_MAXBLOCKINST, "1");
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_MAXBLOCKINST, "5000");
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_SINGLESTEP, "0");
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_MULTIBLOCK, "0");
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_MULTIBLOCK, "1");
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_GDBSERVER, "0");
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_EMULATED_CPU_CORES, "1");
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_ROOTFSPATH, "");
|
||||
|
||||
@@ -1,3 +1,7 @@
|
||||
enable_language(ASM_NASM)
|
||||
if(NOT CMAKE_ASM_NASM_COMPILER_LOADED)
|
||||
error("Failed to find NASM compatible assembler!")
|
||||
endif()
|
||||
|
||||
# Careful. Globbing can't see changes to the contents of files
|
||||
# Need to do a fresh clean to see changes
|
||||
|
||||
@@ -1,3 +1,7 @@
|
||||
enable_language(ASM_NASM)
|
||||
if(NOT CMAKE_ASM_NASM_COMPILER_LOADED)
|
||||
error("Failed to find NASM compatible assembler!")
|
||||
endif()
|
||||
|
||||
# Careful. Globbing can't see changes to the contents of files
|
||||
# Need to do a fresh clean to see changes
|
||||
|
||||
@@ -23,3 +23,14 @@ Test_Primary/Primary_E9.asm
|
||||
Test_X87/D9_F9.asm
|
||||
|
||||
Test_X87/D9_F2.asm
|
||||
|
||||
# Doesn't support unaligned on armv8.0
|
||||
Test_TwoByte/0F_B0_2.asm
|
||||
Test_TwoByte/0F_B0_3.asm
|
||||
Test_TwoByte/0F_B0_4.asm
|
||||
Test_TwoByte/0F_B0_5.asm
|
||||
Test_TwoByte/0F_B0_6.asm
|
||||
Test_TwoByte/0F_B0_7.asm
|
||||
Test_Secondary/09_XX_01_7.asm
|
||||
Test_Secondary/09_XX_01_8.asm
|
||||
Test_Secondary/09_XX_01_9.asm
|
||||
@@ -0,0 +1,21 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"Match": "All",
|
||||
"RegData": {
|
||||
"RAX": "0x20"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov rax, 0
|
||||
cmp rax, 0
|
||||
|
||||
jz finish
|
||||
|
||||
; multiblock should gracefully handle these invalid ops
|
||||
db 0xf, 0x3B ; invalid opcode here
|
||||
|
||||
finish:
|
||||
mov rax, 32
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,39 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x4141414180000000",
|
||||
"RDX": "0x41414141FFFFFFFF",
|
||||
"RBX": "0xFFFFFFFF41424344",
|
||||
"RCX": "0xFFFFFFFF51525354",
|
||||
"R13": "0x5152535441424344",
|
||||
"R14": "0x1"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; Within 16 byte region but unaligned
|
||||
mov r15, 0xe0000007
|
||||
|
||||
mov rax, 0xFFFFFFFF80000000
|
||||
mov [r15 + 8 * 0], rax
|
||||
|
||||
mov r14, 0
|
||||
; Expected
|
||||
mov rax, 0x4141414180000000
|
||||
mov rdx, 0x41414141FFFFFFFF
|
||||
|
||||
; Desired
|
||||
mov rbx, 0xFFFFFFFF41424344
|
||||
mov rcx, 0xFFFFFFFF51525354
|
||||
|
||||
cmpxchg8b [r15]
|
||||
|
||||
; Set r14 to 1 if if the memory location was expected
|
||||
setz r14b
|
||||
|
||||
; Memory will now be set to the register data
|
||||
; EDX:EAX will be the original data
|
||||
|
||||
; Check memory location to ensure it contains what we want
|
||||
mov r13, [r15 + 8 * 0]
|
||||
hlt
|
||||
Loaded 100 of 125 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user