Compare commits

..
2 Commits
Author SHA1 Message Date
Ryan Houdek faed139c34 Docs: Update for release FEX-2312.1 2023-12-11 13:48:42 -08:00
Ryan Houdek 80927bf0e1 FEXLoader: Temporarily disable CLONE_CLEAR_SIGHAND
This flag breaks FEX heavily for now.
glibc 2.38 started using this flag as an optimization for posix_spawn.
It will fall back to a "non-optimized" implementation if the clone
syscall returns EINVAL. For now do this while we investigate a more
proper implementation.

Should be backported to 2312.1.
2023-12-11 13:47:40 -08:00
470 changed files with 43206 additions and 34824 deletions

No files matched your search

@@ -37,6 +37,7 @@ If applicable, add screenshots and video to help explain your problem.
**Additional context**
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
- Is this a Vulkan game: [Yes/No/Unknown]
- If Yes, What is your Vulkan driver:
+14 -1
View File
@@ -13,6 +13,7 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
FEX_ENABLEAVX: 1
jobs:
@@ -77,6 +78,18 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -237,7 +250,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
+14 -1
View File
@@ -20,6 +20,7 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
FEX_ENABLEAVX: 1
jobs:
@@ -84,6 +85,18 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -171,7 +184,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
+1 -1
View File
@@ -90,7 +90,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+1 -1
View File
@@ -121,7 +121,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+13 -1
View File
@@ -100,13 +100,25 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM128bit.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+6 -6
View File
@@ -15,19 +15,19 @@
path = External/tiny-json
url = https://github.com/Sonicadvance1/tiny-json.git
[submodule "External/xbyak"]
shallow = true
shallow = true
path = External/xbyak
url = https://github.com/herumi/xbyak.git
url = https://github.com/FEX-Emu/xbyak.git
[submodule "External/fex-posixtest-bins"]
shallow = true
shallow = true
path = External/fex-posixtest-bins
url = https://github.com/FEX-Emu/fex-posixtest-bins.git
[submodule "External/fex-gvisor-tests-bins"]
shallow = true
shallow = true
path = External/fex-gvisor-tests-bins
url = https://github.com/FEX-Emu/fex-gvisor-tests-bins.git
[submodule "External/fex-gcc-target-tests-bins"]
shallow = true
shallow = true
path = External/fex-gcc-target-tests-bins
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
[submodule "External/jemalloc"]
@@ -41,7 +41,7 @@
url = https://github.com/FEX-Emu/drm-headers.git
[submodule "External/xxhash"]
path = External/xxhash
url = https://github.com/Cyan4973/xxHash.git
url = https://github.com/FEX-Emu/xxHash.git
[submodule "External/Catch2"]
path = External/Catch2
url = https://github.com/catchorg/Catch2.git
+15 -9
View File
@@ -26,6 +26,7 @@ option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
@@ -121,11 +122,6 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
add_definitions(-D_M_ARM_64=1)
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
set(_M_ARM_64EC 1)
add_definitions(-D_M_ARM_64EC=1)
endif()
if (ENABLE_CCACHE)
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
@@ -162,6 +158,18 @@ if (NOT ENABLE_OFFLINE_TELEMETRY)
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
endif()
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
add_definitions(-DTERMUX_BUILD=1)
set(TERMUX_BUILD 1)
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
set(ENABLE_JEMALLOC FALSE)
# Termux builds can't rely on X11 packages
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
set(BUILD_FEXCONFIG FALSE)
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
@@ -224,10 +232,8 @@ endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
set(XXHASH_BUNDLED_MODE TRUE)
set(XXHASH_BUILD_XXHSUM FALSE)
set(BUILD_SHARED_LIBS OFF)
add_subdirectory(External/xxhash/cmake_unofficial/)
add_subdirectory(External/xxhash/)
include_directories(External/xxhash/)
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
+5
View File
@@ -0,0 +1,5 @@
{
"Config": {
"AdditionalArguments": "--no-sandbox"
}
}
+4 -5
View File
@@ -3,12 +3,11 @@ FROM ubuntu:20.04 as builder
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
clang-10 llvm-10 nasm ninja-build pkg-config \
libcap-dev libglfw3-dev libepoxy-dev python3-dev libsdl2-dev \
python3 linux-headers-generic \
git
clang-10 llvm-10 nasm ninja-build \
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
python3 linux-headers-generic
RUN git clone --recurse-submodules https://github.com/FEX-Emu/FEX.git
COPY . /opt/FEX
CMD [ "mkdir /opt/FEX/build" ]
+1 -1
+1 -1
+1 -1
+1 -1
+2 -3
View File
@@ -30,11 +30,10 @@ set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
check_cxx_source_compiles(
"
__attribute__((preserve_all))
int Testy(int a, int b, int c, int d, int e, int f) {
return a + b + c + d + e + f;
void Testy() {
}
int main() {
return Testy(0, 1, 2, 3, 4, 5);
return 0;
}"
HAS_CLANG_PRESERVE_ALL)
unset(CMAKE_REQUIRED_FLAGS)
-21
View File
@@ -441,24 +441,6 @@ def print_parse_envloader_options(options):
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_jsonloader_options(options):
output_argloader.write("#ifdef JSONLOADER\n")
output_argloader.write("#undef JSONLOADER\n")
output_argloader.write("if (false) {}\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
value_type = op_vals["Type"]
if (value_type == "strenum"):
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
output_argloader.write("Set(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
output_argloader.write("}\n")
output_argloader.write("else {{\n".format(op_key))
output_argloader.write("Set(KeyOption, ConfigString);\n")
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_enum_options(options):
output_argloader.write("#ifdef ENUMDEFINES\n")
output_argloader.write("#undef ENUMDEFINES\n")
@@ -574,9 +556,6 @@ print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
# Generate json loader code
print_parse_jsonloader_options(options);
# Generate enum variable options
print_parse_enum_options(options);
+2 -2
View File
@@ -7,7 +7,6 @@ set (FEXCORE_BASE_SRCS
Utils/FileLoading.cpp
Utils/ForcedAssert.cpp
Utils/LogManager.cpp
Utils/SpinWaitLock.cpp
)
if (NOT MINGW_BUILD)
@@ -150,6 +149,7 @@ set (SRCS
Interface/IR/Passes/DeadStoreElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/InlineCallOptimization.cpp
Utils/NetStream.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/Profiler.cpp
@@ -194,7 +194,7 @@ endif()
# Some defines for the softfloat library
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils)
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
if (NOT MINGW_BUILD)
list (APPEND LIBS dl)
+1 -1
View File
@@ -1,10 +1,10 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <cmath>
#include <cstring>
+1 -19
View File
@@ -48,7 +48,6 @@ namespace DefaultValues {
PATH_CONFIG_DIR_GLOBAL,
PATH_CONFIG_FILE_LOCAL,
PATH_CONFIG_FILE_GLOBAL,
PATH_CONFIG_TELEMETRY_FOLDER,
PATH_LAST,
};
static std::array<fextl::string, Paths::PATH_LAST> Paths;
@@ -65,22 +64,6 @@ namespace DefaultValues {
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
}
fextl::string const& GetTelemetryDirectory() {
auto &Path = Paths[PATH_CONFIG_TELEMETRY_FOLDER];
if (Path.empty()) {
FEX_CONFIG_OPT(TelemetryDirectory, TELEMETRYDIRECTORY);
if (!TelemetryDirectory().empty()) {
Path = TelemetryDirectory;
Path += "/";
}
else {
Path = Config::GetDataDirectory() + "Telemetry/";
}
}
return Path;
}
fextl::string const& GetDataDirectory() {
return Paths[PATH_DATA_DIR];
}
@@ -130,7 +113,7 @@ namespace DefaultValues {
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
static FEXCore::Config::Layer *Meta{};
constexpr std::array<FEXCore::Config::LayerType, 10> LoadOrder = {
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
FEXCore::Config::LayerType::LAYER_MAIN,
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP,
@@ -138,7 +121,6 @@ namespace DefaultValues {
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
FEXCore::Config::LayerType::LAYER_ARGUMENTS,
FEXCore::Config::LayerType::LAYER_USER_OVERRIDE,
FEXCore::Config::LayerType::LAYER_ENVIRONMENT,
FEXCore::Config::LayerType::LAYER_TOP
};
+22 -58
View File
@@ -77,9 +77,7 @@
"ENABLECRYPTO": "enablecrypto",
"DISABLECRYPTO": "disablecrypto",
"ENABLERPRES": "enablerpres",
"DISABLERPRES": "disablerpres",
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
"DISABLEPRESERVEALLABI": "disablepreserveallabi"
"DISABLERPRES": "disablerpres"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
@@ -99,28 +97,7 @@
"\t{enable,disable}flagm: Will force enable or disable flagm even if the host doesn't support it",
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it",
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
]
},
"CPUID": {
"Type": "strenum",
"Default": "FEXCore::Config::CPUID::OFF",
"Enums": {
"ENABLESHA": "enablesha",
"DISABLESHA": "disablesha"
},
"Desc": [
"Allows controlling of the CPU features are exposed in CPUID.",
"\toff: Default CPU features queried from CPU features",
"\t{enable,disable}sha: Will force enable or disable sha even if the host doesn't support it"
]
},
"SmallTSCScale": {
"Type": "bool",
"Default": "true",
"Desc": [
"Scales the cycle counter on systems that have low frequencies."
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it"
]
}
},
@@ -270,6 +247,23 @@
"Disables optimizations passes for debugging."
]
},
"SRA": {
"Type": "bool",
"Default": "true",
"Desc": [
"Set to false to disable Static Register Allocation"
]
},
"Force32BitAllocator": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces use of the 32-bit allocator on 32-bit applications",
"Used to work around ulimit problems of CI runner",
"Potentially useful for debugging memory problems",
"32-bit allocator is always used if your host kernel is older than 4.17"
]
},
"GlobalJITNaming": {
"Type": "bool",
"Default": "false",
@@ -368,14 +362,6 @@
"File to write FEX output to.",
"[stdout, stderr, server, <Filename>]"
]
},
"TelemetryDirectory": {
"Type": "str",
"Default": "",
"Desc": [
"Redirects the telemetry folder that FEX usually writes to.",
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
]
}
},
"Hacks": {
@@ -387,8 +373,9 @@
"Desc": [
"Checks code for modification before execution.",
"\tnone: No checks",
"\tmtrack: Page tracking based invalidation (default)",
"\tfull: Validate code before every run (slow)"
"\tmtrack: Page tracking based invalidation",
"\tfull: Validate code before every run (slow)",
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
]
},
"TSOEnabled": {
@@ -399,21 +386,6 @@
"Highly likely to break any multithreaded application if disabled."
]
},
"VectorTSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
]
},
"MemcpySetTSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
"Only affects REP MOVS and REP STOS instructions"
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
@@ -461,14 +433,6 @@
"Hides the hypervisor CPUID bit when set.",
"Should only be used for applications that have issues with this set."
]
},
"StartupSleep": {
"Type": "uint32",
"Default": "0",
"Desc": [
"Sleeps the process at startup for a duration of seconds.",
"Useful if an application crashes too quickly to attach a debugger."
]
}
},
"Misc": {
+12 -4
View File
@@ -12,6 +12,10 @@
#include <string.h>
#include <utility>
namespace FEXCore::HLE {
class SyscallVisitor;
}
namespace FEXCore::Context {
void InitializeStaticTables(OperatingMode Mode) {
X86Tables::InitializeInfoTables(Mode);
@@ -30,6 +34,10 @@ namespace FEXCore::Context {
return CustomExitHandler;
}
void FEXCore::Context::ContextImpl::Stop() {
Stop(false);
}
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
CompileBlock(Thread->CurrentFrame, GuestRIP);
}
@@ -38,6 +46,10 @@ namespace FEXCore::Context {
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
}
bool FEXCore::Context::ContextImpl::IsDone() const {
return IsPaused();
}
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
CustomCPUFactory = std::move(Factory);
}
@@ -66,8 +78,4 @@ namespace FEXCore::Context {
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
return CPUID.RunFunctionName(Function, Leaf, CPU);
}
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState *Thread, uintptr_t Address) const {
return Thread->CPUBackend->IsAddressInCodeBuffer(Address);
}
}
+81 -35
View File
@@ -13,7 +13,6 @@
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
@@ -45,6 +44,7 @@ namespace CodeSerialize {
namespace CPU {
class Arm64JITCore;
class X86JITCore;
class Dispatcher;
}
namespace HLE {
@@ -69,30 +69,28 @@ namespace FEXCore::Context {
MODE_SINGLESTEP = 1,
};
struct ExitFunctionLinkData {
uint64_t HostBranch;
uint64_t GuestRIP;
};
using BlockDelinkerFunc = void(*)(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record);
constexpr uint32_t TSC_SCALE = 128;
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
class ContextImpl final : public FEXCore::Context::Context {
public:
// Context base class implementation.
bool InitCore() override;
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
void SetExitHandler(ExitHandler handler) override;
ExitHandler GetExitHandler() const override;
ExitReason RunUntilExit(FEXCore::Core::InternalThreadState *Thread) override;
void Pause() override;
void Run() override;
void Stop() override;
void Step() override;
ExitReason RunUntilExit() override;
void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) override;
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
bool IsDone() const override;
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
HostFeatures GetHostFeatures() const override;
@@ -119,8 +117,7 @@ namespace FEXCore::Context {
* - CTX->RunUntilExit(Thread);
* OS thread Creation:
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
* - ThreadHandler calls `CTX->ExecutionThread(Thread)`
* - InitializeThread(Thread);
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
@@ -135,13 +132,27 @@ namespace FEXCore::Context {
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
*
* @param Thread The internal FEX thread state object
*
* The OS thread will wait until RunThread is executed
*/
void InitializeThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Starts the OS thread object to start executing guest code
*
* @param Thread The internal FEX thread state object
*/
void RunThread(FEXCore::Core::InternalThreadState *Thread) override;
void StopThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Destroys this FEX thread object and stops tracking it internally
*
* @param Thread The internal FEX thread state object
*/
void DestroyThread(FEXCore::Core::InternalThreadState *Thread, bool NeedsTLSUninstall) override;
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
#ifndef _WIN32
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
@@ -173,22 +184,13 @@ namespace FEXCore::Context {
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
FEXCore::ForkableSharedMutex &GetCodeInvalidationMutex() override {
return CodeInvalidationMutex;
}
void MarkMemoryShared(FEXCore::Core::InternalThreadState *Thread) override;
void MarkMemoryShared() override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState *Thread, uintptr_t Address) const override;
// returns false if a handler was already registered
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr);
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) override;
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
@@ -197,6 +199,9 @@ namespace FEXCore::Context {
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
#ifdef JIT_X86_64
friend class FEXCore::CPU::X86JITCore;
#endif
friend class FEXCore::IR::Validation::IRValidation;
@@ -227,6 +232,7 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
@@ -236,16 +242,25 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
} Config;
FEXCore::HostFeatures HostFeatures;
std::mutex ThreadCreationMutex;
FEXCore::Core::InternalThreadState* ParentThread{};
fextl::vector<FEXCore::Core::InternalThreadState*> Threads;
std::atomic_bool CoreShuttingDown{false};
bool NeedToCheckXID{true};
std::mutex IdleWaitMutex;
std::condition_variable IdleWaitCV;
std::atomic<uint32_t> IdleWaitRefCount{};
Event PauseWait;
bool Running{};
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
FEXCore::HostFeatures HostFeatures;
// CPUID depends on HostFeatures so needs to be initialized after that.
FEXCore::CPUIDEmu CPUID;
FEXCore::HLE::SyscallHandler *SyscallHandler{};
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
@@ -265,15 +280,21 @@ namespace FEXCore::Context {
ContextImpl();
~ContextImpl();
bool IsPaused() const { return !Running; }
void WaitForThreadsToRun() override;
void Stop(bool IgnoreCurrentThread);
void WaitForIdle() override;
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData *HostLink, const BlockDelinkerFunc &delinker);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, ExitFunctionLinkData *Record) {
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
return Fn(Frame, Record);
return Fn(Frame, record);
}
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
@@ -281,7 +302,7 @@ namespace FEXCore::Context {
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
LOGMAN_THROW_A_FMT(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ThreadRemoveCodeEntry(Thread, GuestRIP);
@@ -311,6 +332,9 @@ namespace FEXCore::Context {
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
// same as CompileBlock, but aborts on failure
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
// Used for thread creation from syscalls
/**
* @brief Initializes TID, PID and TLS data for a thread
@@ -335,6 +359,10 @@ namespace FEXCore::Context {
}
}
void IncrementIdleRefCount() override {
++IdleWaitRefCount;
}
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
@@ -364,9 +392,18 @@ namespace FEXCore::Context {
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
ThreadsState GetThreads() override {
return ThreadsState {
.ParentThread = ParentThread,
.Threads = &Threads,
};
}
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
protected:
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
// If the hardware supports TSO then we don't need to emulate it through atomics.
@@ -388,8 +425,15 @@ namespace FEXCore::Context {
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
void WaitForIdleWithTimeout();
void NotifyPause();
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
// Entry Cache
std::mutex ExitMutex;
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
@@ -401,7 +445,9 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
std::shared_mutex CustomIRMutex;
std::atomic<bool> HasCustomIRHandlers{};
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void *, void *>> CustomIRHandlers;
FEXCore::CPU::DispatcherConfig DispatcherConfig;
};
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
}
@@ -9,11 +9,10 @@
#include "Interface/HLE/Thunks/Thunks.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <aarch64/cpu-aarch64.h>
#include <aarch64/instructions-aarch64.h>
#include <cpu-features.h>
@@ -29,7 +28,6 @@ namespace FEXCore::CPU {
// TODO: Allow x18 register allocation on Linux in the future to gain one more register.
namespace x64 {
#ifndef _M_ARM_64EC
// All but x19 and x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
@@ -83,54 +81,6 @@ namespace x64 {
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
};
#else
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r0,
FEXCore::ARMEmitter::Reg::r1, FEXCore::ARMEmitter::Reg::r27,
// SP's register location isn't specified by the ARM64EC ABI, we choose to use r23
FEXCore::ARMEmitter::Reg::r23, FEXCore::ARMEmitter::Reg::r29,
FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r26,
FEXCore::ARMEmitter::Reg::r2, FEXCore::ARMEmitter::Reg::r3,
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r20,
FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22,
REG_PF, REG_AF,
};
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r14,FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r30,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
{FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7},
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
}};
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
};
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
};
#endif
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
@@ -390,7 +340,7 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
, EmitterCTX {ctx}
#ifdef VIXL_SIMULATOR
, Simulator {&SimDecoder, stdout, vixl::aarch64::SimStack(SimulatorStackSize).Allocate()}
, Simulator {&SimDecoder}
#endif
{
#ifdef VIXL_SIMULATOR
@@ -403,10 +353,8 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
// Only setup the disassembler if enabled.
// vixl's decoder is expensive to setup.
if (Disassemble()) {
DisasmBuffer.resize(DISASM_BUFFER_SIZE);
Disasm = fextl::make_unique<vixl::aarch64::Disassembler>(DisasmBuffer.data(), DISASM_BUFFER_SIZE);
DisasmDecoder = fextl::make_unique<vixl::aarch64::Decoder>();
DisasmDecoder->AppendVisitor(Disasm.get());
DisasmDecoder->AppendVisitor(&Disasm);
}
#endif
@@ -419,9 +367,6 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
GeneralPairRegisters = x64::RAPair;
StaticFPRegisters = x64::SRAFPR;
GeneralFPRegisters = x64::RAFPR;
#ifdef _M_ARM_64EC
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
#endif
}
else {
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
@@ -666,6 +611,10 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
if (!StaticRegisterAllocation()) {
return;
}
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
@@ -779,6 +728,10 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
if (!StaticRegisterAllocation()) {
return;
}
if (FPRs) {
// Set up predicate registers.
// We don't bother spilling these in SpillStaticRegs,
@@ -983,9 +936,7 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
// Push the general registers.
PushGeneralRegisters(TmpReg, ConfiguredDynamicRegisterBase);
#ifndef _M_ARM_64EC
str(ARMEmitter::XReg::lr, TmpReg, 0);
#endif
}
void Arm64Emitter::PopDynamicRegsAndLR() {
@@ -997,9 +948,7 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
// Pop GPRs second
PopGeneralRegisters(ConfiguredDynamicRegisterBase);
#ifndef _M_ARM_64EC
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
#endif
}
void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs) {
@@ -21,7 +21,6 @@
#endif
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/vector.h>
#include <array>
#include <cstddef>
@@ -37,37 +36,16 @@ namespace FEXCore::CPU {
// Contains the address to the currently available CPU state
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
#ifndef _M_ARM_64EC
// GPR temporaries. Only x3 can be used across spill boundaries
// so if these ever need to change, be very careful about that.
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x0;
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x1;
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x2;
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x3;
constexpr bool TMP_ABIARGS = true;
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
// Vector temporaries
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v0;
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
#else
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x10;
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x11;
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x12;
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x13;
constexpr bool TMP_ABIARGS = false;
// We pin r11/r12 as PF/AF respectively for arm64ec, as r26/r27 are used for SRA.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r9;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r24;
// Vector temporaries
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v16;
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v17;
#endif
// Predicate register temporaries (used when AVX support is enabled)
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
@@ -75,6 +53,9 @@ constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v17;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
@@ -249,17 +230,15 @@ protected:
#ifdef VIXL_SIMULATOR
vixl::aarch64::Decoder SimDecoder;
vixl::aarch64::Simulator Simulator;
constexpr static size_t SimulatorStackSize = 8 * 1024 * 1024;
#endif
#ifdef VIXL_DISASSEMBLER
fextl::vector<char> DisasmBuffer;
constexpr static int DISASM_BUFFER_SIZE {256};
fextl::unique_ptr<vixl::aarch64::Disassembler> Disasm;
vixl::aarch64::Disassembler Disasm;
fextl::unique_ptr<vixl::aarch64::Decoder> DisasmDecoder;
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
#endif
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
};
}
@@ -35,10 +35,8 @@ public:
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void adr(FEXCore::ARMEmitter::Register rd, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADR });
void adr(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADR });
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
}
@@ -64,10 +62,8 @@ public:
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void adrp(FEXCore::ARMEmitter::Register rd, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADRP });
void adrp(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADRP });
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
}
@@ -109,7 +105,7 @@ public:
}
}
void LongAddressGen(FEXCore::ARMEmitter::Register rd, ForwardLabel* Label) {
Label->Insts.emplace_back(SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN });
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN });
// Emit a register index and a nop. These will be backpatched.
dc32(rd.Idx());
nop();
@@ -18,10 +18,8 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void b(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
}
@@ -47,10 +45,8 @@ public:
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bc(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void bc(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
}
@@ -106,10 +102,8 @@ public:
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
void b(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
@@ -137,10 +131,8 @@ public:
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bl(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
void bl(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
@@ -171,10 +163,8 @@ public:
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0100 << 24;
@@ -205,10 +195,8 @@ public:
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0101 << 24;
@@ -238,11 +226,8 @@ public:
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0110 << 24;
@@ -271,11 +256,8 @@ public:
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
@@ -4,14 +4,13 @@
#include "Interface/Core/ArchHelpers/CodeEmitter/Buffer.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <aarch64/assembler-aarch64.h>
#include <array>
@@ -538,29 +537,27 @@ namespace FEXCore::ARMEmitter {
uint8_t *Location{};
};
/* This `SingleUseForwardLabel` struct used for retaining a location for PC-Relative instructions.
/* This `ForwardLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is logically `above` an instruction that uses it.
* Which means that a branch would jump forwards.
*
* The `ForwardLabel` struct can be bound to multiple instructions, so it needs a vector for each bind instruction type.
* This can be bound to multiple instructions, so it needs a vector for each bind instruction type.
*/
struct SingleUseForwardLabel {
enum class InstType {
UNKNOWN,
ADR,
ADRP,
B,
BC,
TEST_BRANCH,
RELATIVE_LOAD,
LONG_ADDRESS_GEN,
};
uint8_t *Location{};
InstType Type = InstType::UNKNOWN;
};
struct ForwardLabel {
fextl::vector<SingleUseForwardLabel> Insts{};
struct Instructions {
enum class InstType {
ADR,
ADRP,
B,
BC,
TEST_BRANCH,
RELATIVE_LOAD,
LONG_ADDRESS_GEN,
};
uint8_t *Location{};
InstType Type;
};
fextl::vector<Instructions> Insts{};
};
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
@@ -572,15 +569,6 @@ namespace FEXCore::ARMEmitter {
ForwardLabel Forward;
};
static inline void AddLocationToLabel(SingleUseForwardLabel *Label, SingleUseForwardLabel&& Location) {
LOGMAN_THROW_A_FMT(Label->Type == SingleUseForwardLabel::InstType::UNKNOWN, "Trying to bind a SingleUseForwardLabel to multiple locations. Use ForwardLabel instead.");
*Label = std::move(Location);
}
static inline void AddLocationToLabel(ForwardLabel *Label, SingleUseForwardLabel&& Location) {
Label->Insts.emplace_back(std::move(Location));
}
// Some FCMA ASIMD instructions support a rotation argument.
enum class Rotation : uint32_t {
ROTATE_0 = 0b00,
@@ -640,121 +628,6 @@ namespace FEXCore::ARMEmitter {
Label->Location = GetCursorAddress<uint8_t*>();
}
void Bind(const SingleUseForwardLabel *Label) {
uint8_t *CurrentAddress = GetCursorAddress<uint8_t*>();
// Patch up the instructions
switch (Label->Type) {
case SingleUseForwardLabel::InstType::ADR: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::ADRP: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::B: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= Offset;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::TEST_BRANCH: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::BC:
case SingleUseForwardLabel::InstType::RELATIVE_LOAD: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN: {
uint32_t *Instructions = reinterpret_cast<uint32_t*>(Label->Location);
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
auto OriginalOffset = GetCursorOffset();
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
SetCursorOffset(InstOffset);
// We encoded the destination register in to the first instruction space.
// Read it back.
ARMEmitter::Register DestReg(Instructions[0]);
if (IsADRRange(ImmInstTwo)) {
// If within ADR range from the second instruction, then we can emit NOP+ADR
nop();
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
}
else if (IsADRPRange(ImmInstOne)) {
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
// First check if we are in non-offset range for second instruction.
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
// We can emit nop + adrp
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
}
else {
// Not aligned, need adrp + add
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
}
}
else {
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
FEX_UNREACHABLE;
}
SetCursorOffset(OriginalOffset);
break;
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
}
// Bind a forward label to a location.
// This walks all the instructions in the label's vector.
// Then backpatching all instructions that have used the label.
@@ -763,8 +636,119 @@ namespace FEXCore::ARMEmitter {
if constexpr (WarnAboutEmpty) {
LOGMAN_THROW_A_FMT(Label->Insts.empty() == false, "Binding forward label that didn't have any instructions using it");
}
for (auto &Inst : Label->Insts) {
Bind(&Inst);
uint8_t *CurrentAddress = GetCursorAddress<uint8_t*>();
for (const auto &Inst : Label->Insts) {
// Patch up the instructions
switch (Inst.Type) {
case ForwardLabel::Instructions::InstType::ADR: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::ADRP: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::B: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= Offset;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::TEST_BRANCH: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::BC:
case ForwardLabel::Instructions::InstType::RELATIVE_LOAD: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN: {
uint32_t *Instructions = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
auto OriginalOffset = GetCursorOffset();
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
SetCursorOffset(InstOffset);
// We encoded the destination register in to the first instruction space.
// Read it back.
ARMEmitter::Register DestReg(Instructions[0]);
if (IsADRRange(ImmInstTwo)) {
// If within ADR range from the second instruction, then we can emit NOP+ADR
nop();
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
}
else if (IsADRPRange(ImmInstOne)) {
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
// First check if we are in non-offset range for second instruction.
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
// We can emit nop + adrp
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
}
else {
// Not aligned, need adrp + add
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
}
}
else {
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
FEX_UNREACHABLE;
}
SetCursorOffset(OriginalOffset);
break;
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
}
}
@@ -2121,58 +2121,38 @@ public:
LoadStoreLiteral(Op, prfop, static_cast<uint32_t>(Imm >> 2) & 0x7'FFFF);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::WRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::WRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b0001'1000 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::SRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::SRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b0001'1100 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::XRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::XRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b0101'1000 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::DRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::DRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b0101'1100 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldrsw(FEXCore::ARMEmitter::XRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldrsw(FEXCore::ARMEmitter::XRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b1001'1000 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::QRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::QRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b1001'1100 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void prfm(FEXCore::ARMEmitter::Prefetch prfop, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void prfm(FEXCore::ARMEmitter::Prefetch prfop, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b1101'1000 << 24;
LoadStoreLiteral(Op, prfop, 0);
}
@@ -3762,12 +3742,7 @@ public:
}
else {
if (MemSrc.MetaType.ImmType.Index == ARMEmitter::IndexType::OFFSET) {
if ((MemSrc.MetaType.ImmType.Imm & 0b111) || MemSrc.MetaType.ImmType.Imm < 0) {
prfum<IndexType::OFFSET>(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
}
else {
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
}
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
}
else {
LOGMAN_MSG_A_FMT("Unexpected loadstore index type");
+1 -32
View File
@@ -2,12 +2,8 @@
#include "FEXCore/IR/IR.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#ifndef _WIN32
#include <sys/prctl.h>
#endif
#include <FEXCore/Core/CPUBackend.h>
namespace FEXCore {
namespace CPU {
@@ -27,8 +23,6 @@ constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant:
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB_UPPER
};
constexpr static auto PSHUFLW_LUT {
@@ -355,31 +349,6 @@ auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
}
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
#ifndef _WIN32
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
//
// MDWE prevents applications from creating RWX memory mappings.
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
//
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
//
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
//
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
// -1: The kernel doesn't support MDWE
// 0: MDWE is supported but disabled
// >0: MDWE is enabled, hence prohibiting RWX mappings
#ifndef PR_GET_MDWE
#define PR_GET_MDWE 66
#endif
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
if (MDWE != -1 && MDWE != 0) {
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
}
#endif
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t *>(
+11 -36
View File
@@ -98,7 +98,7 @@ constexpr uint32_t FAMILY_IDENTIFIER =
#endif
#ifdef _M_ARM_64
uint32_t GetCycleCounterFrequency() {
static uint32_t GetCycleCounterFrequency() {
uint64_t Result{};
__asm("mrs %[Res], CNTFRQ_EL0"
: [Res] "=r" (Result));
@@ -349,7 +349,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
}
#else
uint32_t GetCycleCounterFrequency() {
static uint32_t GetCycleCounterFrequency() {
return 0;
}
@@ -358,34 +358,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
#endif
void CPUIDEmu::SetupFeatures() {
// TODO: Enable once AVX is supported.
if (false && CTX->HostFeatures.SupportsAVX) {
XCR0 |= XCR0_AVX;
}
// Override features if the user has specifically called for it.
FEX_CONFIG_OPT(CPUIDFeatures, CPUID);
if (!CPUIDFeatures()) {
// Early exit if no features are overriden.
return;
}
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
do { \
const bool Disable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::DISABLE##enum_name) != 0; \
const bool Enable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::ENABLE##enum_name) != 0; \
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
const bool AlreadyEnabled = Features.FeatureName; \
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
Features.FeatureName = Result; \
} while (0)
ENABLE_DISABLE_OPTION(SHA, SHA, SHA);
#undef ENABLE_DISABLE_OPTION
}
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
FEXCore::CPUID::FunctionResults Res{};
@@ -667,7 +639,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 26) | // Reserved
(0 << 27) | // Reserved
(0 << 28) | // Reserved
(Features.SHA << 29) | // SHA instructions
(1 << 29) | // SHA instructions
(0 << 30) | // Reserved
(0 << 31); // Reserved
@@ -694,7 +666,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 19) | // MPX MAWAU
(0 << 20) | // MPX MAWAU
(0 << 21) | // MPX MAWAU
(1 << 22) | // RDPID Read Processor ID
(0 << 22) | // RDPID Read Processor ID
(0 << 23) | // Reserved
(0 << 24) | // Reserved
(0 << 25) | // CLDEMOTE
@@ -803,7 +775,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) const {
uint32_t FrequencyHz = GetCycleCounterFrequency();
if (FrequencyHz) {
Res.eax = 1;
Res.ebx = CTX->Config.SmallTSCScale() ? FEXCore::Context::TSC_SCALE : 1;
Res.ebx = 1;
Res.ecx = FrequencyHz;
}
return Res;
@@ -1233,14 +1205,17 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
return Res;
}
CPUIDEmu::CPUIDEmu(FEXCore::Context::ContextImpl const *ctx)
: CTX {ctx} {
void CPUIDEmu::Init(FEXCore::Context::ContextImpl *ctx) {
CTX = ctx;
Cores = FEXCore::CPUInfo::CalculateNumberOfCPUs();
// Setup some state tracking
SetupHostHybridFlag();
SetupFeatures();
// TODO: Enable once AVX is supported.
if (false && CTX->HostFeatures.SupportsAVX) {
XCR0 |= XCR0_AVX;
}
}
}
+3 -16
View File
@@ -14,8 +14,6 @@ namespace Context {
class ContextImpl;
}
uint32_t GetCycleCounterFrequency();
// Debugging define to switch what family of CPU we execute as.
// Might be useful if an application makes an assumption about a CPU.
// #define CPUID_AMD
@@ -30,12 +28,12 @@ private:
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
public:
CPUIDEmu(FEXCore::Context::ContextImpl const *ctx);
// X86 cacheline size effectively has to be hardcoded to 64
// if we report anything differently then applications are likely to break
constexpr static uint64_t CACHELINE_SIZE = 64;
void Init(FEXCore::Context::ContextImpl *ctx);
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) const {
if (Function < Primary.size()) {
const auto Handler = Primary[Function];
@@ -113,11 +111,10 @@ public:
}
private:
FEXCore::Context::ContextImpl const *CTX;
FEXCore::Context::ContextImpl *CTX;
bool Hybrid{};
uint32_t Cores{};
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
// XFEATURE_ENABLED_MASK
// Mask that configures what features are enabled on the CPU.
@@ -139,15 +136,6 @@ private:
constexpr static uint64_t XCR0_SSE = 1ULL << 1;
constexpr static uint64_t XCR0_AVX = 1ULL << 2;
struct FeaturesConfig {
uint64_t SHA : 1;
uint64_t _pad : 63;
};
FeaturesConfig Features {
.SHA = 1,
};
uint64_t XCR0 {
XCR0_X87 |
XCR0_SSE
@@ -201,7 +189,6 @@ private:
FEXCore::CPUID::XCRResults XCRFunction_0h() const;
void SetupHostHybridFlag();
void SetupFeatures();
static constexpr size_t PRIMARY_FUNCTION_COUNT = 27;
static constexpr size_t HYPERVISOR_FUNCTION_COUNT = 2;
static constexpr size_t EXTENDED_FUNCTION_COUNT = 32;
+373 -64
View File
@@ -12,7 +12,6 @@ $end_info$
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
@@ -21,24 +20,27 @@ $end_info$
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include "Interface/HLE/Thunks/Thunks.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Utils/Allocator.h"
#include "Utils/Allocator/HostAllocator.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CodeLoader.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/HLE/SourcecodeResolver.h>
#include <FEXCore/HLE/Linux/ThreadManagement.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/File.h>
@@ -76,8 +78,7 @@ $end_info$
namespace FEXCore::Context {
ContextImpl::ContextImpl()
: CPUID {this}
, IRCaptureCache {this} {
: IRCaptureCache {this} {
#ifdef BLOCKSTATS
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
#endif
@@ -96,19 +97,32 @@ namespace FEXCore::Context {
Symbols.InitFile();
}
if (FEXCore::GetCycleCounterFrequency() >= FEXCore::Context::TSC_SCALE_MAXIMUM) {
Config.SmallTSCScale = false;
}
// Track atomic TSO emulation configuration.
UpdateAtomicTSOEmulationConfig();
CPUID.Init(this);
}
ContextImpl::~ContextImpl() {
if (ParentThread) {
DestroyThread(ParentThread);
}
{
if (CodeObjectCacheService) {
CodeObjectCacheService->Shutdown();
}
for (auto &Thread : Threads) {
if (Thread->ExecutionThread->joinable()) {
Thread->ExecutionThread->join(nullptr);
}
}
for (auto &Thread : Threads) {
delete Thread;
}
Threads.clear();
}
}
@@ -164,7 +178,6 @@ namespace FEXCore::Context {
case X86State::RFLAG_ZF_RAW_LOC:
case X86State::RFLAG_SF_RAW_LOC:
case X86State::RFLAG_OF_RAW_LOC:
case X86State::RFLAG_DF_RAW_LOC:
// Intentionally do nothing.
// These contain multiple bits which can corrupt other members when compacted.
break;
@@ -213,11 +226,6 @@ namespace FEXCore::Context {
uint32_t AF = ((Frame->State.af_raw ^ PFByte) & (1 << 4)) ? 1 : 0;
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
// DF is pretransformed, undo the transform from 1/-1 back to 0/1
uint8_t DFByte = Frame->State.flags[X86State::RFLAG_DF_RAW_LOC];
if (DFByte & 0x80)
EFLAGS |= 1 << X86State::RFLAG_DF_RAW_LOC;
return EFLAGS;
}
@@ -241,10 +249,6 @@ namespace FEXCore::Context {
// PF is inverted in our internal representation.
Frame->State.pf_raw = (EFLAGS & (1U << i)) ? 0 : 1;
break;
case X86State::RFLAG_DF_RAW_LOC:
// DF is encoded as 1/-1
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 0xff : 1;
break;
default:
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 1 : 0;
break;
@@ -265,7 +269,7 @@ namespace FEXCore::Context {
Frame->State.flags[X86State::RFLAG_IF_LOC] = 1;
}
bool ContextImpl::InitCore() {
FEXCore::Core::InternalThreadState* ContextImpl::InitCore(uint64_t InitialRIP, uint64_t StackPointer) {
// Initialize the CPU core signal handlers & DispatcherConfig
switch (Config.Core) {
case FEXCore::Config::CONFIG_IRJIT:
@@ -275,20 +279,21 @@ namespace FEXCore::Context {
// Do nothing
break;
default:
LogMan::Msg::EFmt("Unknown core configuration");
return false;
ERROR_AND_DIE_FMT("Unknown core configuration");
break;
}
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
DispatcherConfig.StaticRegisterAllocation = Config.StaticRegisterAllocation && BackendFeatures.SupportsStaticRegisterAllocation;
Dispatcher = FEXCore::CPU::Dispatcher::Create(this, DispatcherConfig);
// Set up the SignalDelegator config since core is initialized.
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
.StaticRegisterAllocation = DispatcherConfig.StaticRegisterAllocation,
.SupportsAVX = HostFeatures.SupportsAVX,
.DispatcherBegin = Dispatcher->Start,
.DispatcherEnd = Dispatcher->End,
.AbsoluteLoopTopAddress = Dispatcher->AbsoluteLoopTopAddress,
.AbsoluteLoopTopAddressFillSRA = Dispatcher->AbsoluteLoopTopAddressFillSRA,
.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress,
.SignalHandlerReturnAddressRT = Dispatcher->SignalHandlerReturnAddressRT,
@@ -326,17 +331,174 @@ namespace FEXCore::Context {
StartPaused = true;
}
return true;
FEXCore::Core::InternalThreadState *Thread = CreateThread(InitialRIP, StackPointer, nullptr, 0);
// We are the parent thread
ParentThread = Thread;
return Thread;
}
void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
}
FEXCore::Context::ExitReason ContextImpl::RunUntilExit(FEXCore::Core::InternalThreadState *Thread) {
ExecutionThread(Thread);
void ContextImpl::WaitForIdle() {
std::unique_lock<std::mutex> lk(IdleWaitMutex);
IdleWaitCV.wait(lk, [this] {
return IdleWaitRefCount.load() == 0;
});
Running = false;
}
void ContextImpl::WaitForIdleWithTimeout() {
std::unique_lock<std::mutex> lk(IdleWaitMutex);
bool WaitResult = IdleWaitCV.wait_for(lk, std::chrono::milliseconds(1500),
[this] {
return IdleWaitRefCount.load() == 0;
});
if (!WaitResult) {
// The wait failed, this will occur if we stepped in to a syscall
// That's okay, we just need to pause the threads manually
NotifyPause();
}
// We have sent every thread a pause signal
// Now wait again because they /will/ be going to sleep
WaitForIdle();
}
void ContextImpl::NotifyPause() {
// Tell all the threads that they should pause
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
for (auto &Thread : Threads) {
SignalDelegation->SignalThread(Thread, FEXCore::Core::SignalEvent::Pause);
}
}
void ContextImpl::Pause() {
// If we aren't running, WaitForIdle will never compete.
if (Running) {
NotifyPause();
WaitForIdle();
}
}
void ContextImpl::Run() {
// Spin up all the threads
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
for (auto &Thread : Threads) {
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Return);
}
for (auto &Thread : Threads) {
Thread->StartRunning.NotifyAll();
}
}
void ContextImpl::WaitForThreadsToRun() {
size_t NumThreads{};
{
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
NumThreads = Threads.size();
}
// Spin while waiting for the threads to start up
std::unique_lock<std::mutex> lk(IdleWaitMutex);
IdleWaitCV.wait(lk, [this, NumThreads] {
return IdleWaitRefCount.load() >= NumThreads;
});
Running = true;
}
void ContextImpl::Step() {
{
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
// Walk the threads and tell them to clear their caches
// Useful when our block size is set to a large number and we need to step a single instruction
for (auto &Thread : Threads) {
ClearCodeCache(Thread);
}
}
CoreRunningMode PreviousRunningMode = this->Config.RunningMode;
int64_t PreviousMaxIntPerBlock = this->Config.MaxInstPerBlock;
this->Config.RunningMode = FEXCore::Context::CoreRunningMode::MODE_SINGLESTEP;
this->Config.MaxInstPerBlock = 1;
Run();
WaitForThreadsToRun();
WaitForIdle();
this->Config.RunningMode = PreviousRunningMode;
this->Config.MaxInstPerBlock = PreviousMaxIntPerBlock;
}
void ContextImpl::Stop(bool IgnoreCurrentThread) {
pid_t tid = FHU::Syscalls::gettid();
FEXCore::Core::InternalThreadState* CurrentThread{};
// Tell all the threads that they should stop
{
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
for (auto &Thread : Threads) {
if (IgnoreCurrentThread &&
Thread->ThreadManager.TID == tid) {
// If we are callign stop from the current thread then we can ignore sending signals to this thread
// This means that this thread is already gone
continue;
}
else if (Thread->ThreadManager.TID == tid) {
// We need to save the current thread for last to ensure all threads receive their stop signals
CurrentThread = Thread;
continue;
}
if (Thread->RunningEvents.Running.load()) {
StopThread(Thread);
}
// If the thread is waiting to start but immediately killed then there can be a hang
// This occurs in the case of gdb attach with immediate kill
if (Thread->RunningEvents.WaitingToStart.load()) {
Thread->RunningEvents.EarlyExit = true;
Thread->StartRunning.NotifyAll();
}
}
}
// Stop the current thread now if we aren't ignoring it
if (CurrentThread) {
StopThread(CurrentThread);
}
}
void ContextImpl::StopThread(FEXCore::Core::InternalThreadState *Thread) {
if (Thread->RunningEvents.Running.exchange(false)) {
SignalDelegation->SignalThread(Thread, FEXCore::Core::SignalEvent::Stop);
}
}
void ContextImpl::SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event) {
if (Thread->RunningEvents.Running.load()) {
SignalDelegation->SignalThread(Thread, Event);
}
}
FEXCore::Context::ExitReason ContextImpl::RunUntilExit() {
if(!StartPaused) {
// We will only have one thread at this point, but just in case run notify everything
std::lock_guard lk(ThreadCreationMutex);
for (auto &Thread : Threads) {
Thread->StartRunning.NotifyAll();
}
}
ExecutionThread(ParentThread);
while(true) {
auto reason = Thread->ExitReason;
this->WaitForIdle();
auto reason = ParentThread->ExitReason;
// Don't return if a custom exit handling the exit
if (!CustomExitHandler || reason == ExitReason::EXIT_SHUTDOWN) {
@@ -349,18 +511,61 @@ namespace FEXCore::Context {
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
}
struct ExecutionThreadHandler {
ContextImpl *This;
FEXCore::Core::InternalThreadState *Thread;
};
static void *ThreadHandler(void* Data) {
ExecutionThreadHandler *Handler = reinterpret_cast<ExecutionThreadHandler*>(Data);
Handler->This->ExecutionThread(Handler->Thread);
FEXCore::Allocator::free(Handler);
return nullptr;
}
void ContextImpl::InitializeThread(FEXCore::Core::InternalThreadState *Thread) {
// This will create the execution thread but it won't actually start executing
ExecutionThreadHandler *Arg = reinterpret_cast<ExecutionThreadHandler*>(FEXCore::Allocator::malloc(sizeof(ExecutionThreadHandler)));
Arg->This = this;
Arg->Thread = Thread;
Thread->StartPaused = NeedToCheckXID;
Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
// Wait for the thread to have started
Thread->ThreadWaiting.Wait();
if (NeedToCheckXID) {
// The first time an application creates a thread, GLIBC installs their SETXID signal handler.
// FEX needs to capture all signals and defer them to the guest.
// Once FEX creates its first guest thread, overwrite the GLIBC SETXID handler *again* to ensure
// FEX maintains control of the signal handler on this signal.
NeedToCheckXID = false;
SignalDelegation->CheckXIDHandler();
Thread->StartRunning.NotifyAll();
}
}
void ContextImpl::InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread) {
// Let's do some initial bookkeeping here
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
Thread->ThreadManager.PID = ::getpid();
if (Config.BlockJITNaming() ||
Config.GlobalJITNaming() ||
Config.LibraryJITNaming()) {
// Allocate a TLS JIT symbol buffer only if enabled.
Thread->SymbolBuffer = JITSymbols::AllocateBuffer();
}
SignalDelegation->RegisterTLSState(Thread);
if (ThunkHandler) {
ThunkHandler->RegisterTLSState(Thread);
}
#ifndef _WIN32
Alloc::OSAllocator::RegisterTLSData(Thread);
#endif
}
void ContextImpl::RunThread(FEXCore::Core::InternalThreadState *Thread) {
// Tell the thread to start executing
Thread->StartRunning.NotifyAll();
}
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
@@ -369,6 +574,9 @@ namespace FEXCore::Context {
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(this);
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
Thread->PassManager->RegisterExitHandler([this]() {
Stop(false /* Ignore current thread */);
});
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
@@ -377,7 +585,9 @@ namespace FEXCore::Context {
Thread->CTX = this;
Thread->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT);
bool DoSRA = DispatcherConfig.StaticRegisterAllocation;
Thread->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT, DoSRA);
Thread->PassManager->AddDefaultValidationPasses();
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
@@ -385,7 +595,7 @@ namespace FEXCore::Context {
// Create CPU backend
switch (Config.Core) {
case FEXCore::Config::CONFIG_IRJIT:
Thread->PassManager->InsertRegisterAllocationPass(HostFeatures.SupportsAVX);
Thread->PassManager->InsertRegisterAllocationPass(DoSRA, HostFeatures.SupportsAVX);
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
break;
case FEXCore::Config::CONFIG_CUSTOM:
@@ -419,21 +629,30 @@ namespace FEXCore::Context {
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
if (Config.BlockJITNaming() ||
Config.GlobalJITNaming() ||
Config.LibraryJITNaming()) {
// Allocate a JIT symbol buffer only if enabled.
Thread->SymbolBuffer = JITSymbols::AllocateBuffer();
// Insert after the Thread object has been fully initialized
{
std::lock_guard lk(ThreadCreationMutex);
Threads.push_back(Thread);
}
return Thread;
}
void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState *Thread, bool NeedsTLSUninstall) {
if (NeedsTLSUninstall) {
#ifndef _WIN32
Alloc::OSAllocator::UninstallTLSData(Thread);
#endif
void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState *Thread) {
// remove new thread object
{
std::lock_guard lk(ThreadCreationMutex);
auto It = std::find(Threads.begin(), Threads.end(), Thread);
LOGMAN_THROW_A_FMT(It != Threads.end(), "Thread wasn't in Threads");
Threads.erase(It);
}
if (Thread->ExecutionThread &&
Thread->ExecutionThread->IsSelf()) {
// To be able to delete a thread from itself, we need to detached the std::thread object
Thread->ExecutionThread->detach();
}
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(Thread->CurrentFrame->State.DeferredSignalFaultAddress), 4096);
@@ -451,6 +670,43 @@ namespace FEXCore::Context {
CodeInvalidationMutex.unlock();
return;
}
// This function is called after fork
// We need to cleanup some of the thread data that is dead
for (auto &DeadThread : Threads) {
if (DeadThread == LiveThread) {
continue;
}
// Setting running to false ensures that when they are shutdown we won't send signals to kill them
DeadThread->RunningEvents.Running = false;
// Despite what google searches may susgest, glibc actually has special code to handle forks
// with multiple active threads.
// It cleans up the stacks of dead threads and marks them as terminated.
// It also cleans up a bunch of internal mutexes.
// FIXME: TLS is probally still alive. Investigate
// Deconstructing the Interneal thread state should clean up most of the state.
// But if anything on the now deleted stack is holding a refrence to the heap, it will be leaked
delete DeadThread;
// FIXME: Make sure sure nothing gets leaked via the heap. Ideas:
// * Make sure nothing is allocated on the heap without ref in InternalThreadState
// * Surround any code that heap allocates with a per-thread mutex.
// Before forking, the the forking thread can lock all thread mutexes.
}
// Remove all threads but the live thread from Threads
Threads.clear();
Threads.push_back(LiveThread);
// We now only have one thread
IdleWaitRefCount = 1;
// Clean up dead stacks
FEXCore::Threads::Thread::CleanupAfterFork();
}
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
@@ -475,6 +731,7 @@ namespace FEXCore::Context {
Thread->LookupCache->ClearCache();
Thread->CPUBackend->ClearCache();
Thread->DebugStore.clear();
}
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, IR::IREmitter *IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
@@ -518,20 +775,17 @@ namespace FEXCore::Context {
uint64_t TotalInstructions {0};
uint64_t TotalInstructionsLength {0};
bool HasCustomIR{};
if (HasCustomIRHandlers.load(std::memory_order_relaxed)) {
std::shared_lock lk(CustomIRMutex);
auto Handler = CustomIRHandlers.find(GuestRIP);
if (Handler != CustomIRHandlers.end()) {
TotalInstructions = 1;
TotalInstructionsLength = 1;
std::get<0>(Handler->second)(GuestRIP, Thread->OpDispatcher.get());
HasCustomIR = true;
}
}
std::shared_lock lk(CustomIRMutex);
if (!HasCustomIR) {
auto Handler = CustomIRHandlers.find(GuestRIP);
if (Handler != CustomIRHandlers.end()) {
TotalInstructions = 1;
TotalInstructionsLength = 1;
std::get<0>(Handler->second)(GuestRIP, Thread->OpDispatcher.get());
lk.unlock();
} else {
lk.unlock();
uint8_t const *GuestCode{};
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
@@ -773,6 +1027,17 @@ namespace FEXCore::Context {
};
}
void ContextImpl::CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
auto NewBlock = CompileBlock(Frame, GuestRIP);
if (NewBlock == 0) {
LogMan::Msg::EFmt("CompileBlockJit: Failed to compile code {:X} - aborting process", GuestRIP);
// Return similar behaviour of SIGILL abort
Frame->Thread->StatusCode = 128 + SIGILL;
Stop(false /* Ignore current thread */);
}
}
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst) {
FEXCORE_PROFILE_SCOPED("CompileBlock");
auto Thread = Frame->Thread;
@@ -877,11 +1142,16 @@ namespace FEXCore::Context {
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
InitializeThreadTLSData(Thread);
#ifndef _WIN32
Alloc::OSAllocator::RegisterTLSData(Thread);
#endif
++IdleWaitRefCount;
// Now notify the thread that we are initialized
Thread->ThreadWaiting.NotifyAll();
if (StartPaused || Thread->StartPaused) {
if (Thread != static_cast<ContextImpl*>(Thread->CTX)->ParentThread || StartPaused || Thread->StartPaused) {
// Parent thread doesn't need to wait to run
Thread->StartRunning.Wait();
}
@@ -916,9 +1186,18 @@ namespace FEXCore::Context {
}
}
--IdleWaitRefCount;
IdleWaitCV.notify_all();
#ifndef _WIN32
Alloc::OSAllocator::UninstallTLSData(Thread);
#endif
SignalDelegation->UninstallTLSState(Thread);
// If the parent thread is waiting to join, then we can't destroy our thread object
if (!Thread->DestroyedByParent && Thread != static_cast<ContextImpl*>(Thread->CTX)->ParentThread) {
Thread->CTX->DestroyThread(Thread);
}
}
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
@@ -935,29 +1214,55 @@ namespace FEXCore::Context {
}
}
static void InvalidateGuestCodeRangeInternal(ContextImpl *CTX, uint64_t Start, uint64_t Length) {
std::lock_guard lk(static_cast<ContextImpl*>(CTX)->ThreadCreationMutex);
for (auto &Thread : static_cast<ContextImpl*>(CTX)->Threads) {
InvalidateGuestThreadCodeRange(Thread, Start, Length);
}
}
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
InvalidateGuestThreadCodeRange(Thread, Start, Length);
// Potential deferred since Thread might not be valid.
// Thread object isn't valid very early in frontend's initialization.
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
auto lk = GuardSignalDeferringSectionWithFallback(CodeInvalidationMutex, Thread);
InvalidateGuestCodeRangeInternal(this, Start, Length);
}
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn CallAfter) {
InvalidateGuestThreadCodeRange(Thread, Start, Length);
// Potential deferred since Thread might not be valid.
// Thread object isn't valid very early in frontend's initialization.
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
auto lk = GuardSignalDeferringSectionWithFallback(CodeInvalidationMutex, Thread);
InvalidateGuestCodeRangeInternal(this, Start, Length);
CallAfter(Start, Length);
}
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState *Thread) {
void ContextImpl::MarkMemoryShared() {
if (!IsMemoryShared) {
IsMemoryShared = true;
UpdateAtomicTSOEmulationConfig();
if (Config.TSOAutoMigration) {
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
LogMan::Throw::AFmt(Threads.size() == 1, "First MarkMemoryShared called must be before creating any threads");
auto Thread = Threads[0];
// Only the lookup cache is cleared here, so that old code can keep running until next compilation
std::lock_guard<std::recursive_mutex> lkLookupCache(Thread->LookupCache->WriteLock);
Thread->LookupCache->ClearCache();
// DebugStore also needs to be cleared
Thread->DebugStore.clear();
}
}
}
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData *HostLink, const FEXCore::Context::BlockDelinkerFunc &delinker) {
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
@@ -968,7 +1273,8 @@ namespace FEXCore::Context {
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
Thread->DebugStore.erase(GuestRIP);
Thread->LookupCache->Erase(GuestRIP);
}
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator, void *Data) {
@@ -977,7 +1283,6 @@ namespace FEXCore::Context {
std::unique_lock lk(CustomIRMutex);
auto InsertedIterator = CustomIRHandlers.emplace(Entrypoint, std::tuple(Handler, Creator, Data));
HasCustomIRHandlers = true;
if (!InsertedIterator.second) {
const auto &[fn, Creator, Data] = InsertedIterator.first->second;
@@ -996,8 +1301,12 @@ namespace FEXCore::Context {
InvalidateGuestCodeRange(nullptr, Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
CustomIRHandlers.erase(Entrypoint);
});
}
HasCustomIRHandlers = !CustomIRHandlers.empty();
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args) {
uint64_t Result{};
Result = Handler->HandleSyscall(Frame, Args);
return Result;
}
IR::AOTIRCacheEntry *ContextImpl::LoadAOTIRCacheEntry(const fextl::string &filename) {
@@ -5,14 +5,12 @@
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/X86HelperGen.h"
#include "Utils/MemberFunctionToPointer.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
@@ -25,15 +23,42 @@
namespace FEXCore::CPU {
static void SleepThread(FEXCore::Context::ContextImpl *CTX, FEXCore::Core::CpuStateFrame *Frame) {
CTX->SyscallHandler->SleepThread(CTX, Frame);
void Dispatcher::SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame) {
auto Thread = Frame->Thread;
--ctx->IdleWaitRefCount;
ctx->IdleWaitCV.notify_all();
Thread->RunningEvents.ThreadSleeping = true;
// Go to sleep
Thread->StartRunning.Wait();
Thread->RunningEvents.Running = true;
++ctx->IdleWaitRefCount;
Thread->RunningEvents.ThreadSleeping = false;
ctx->IdleWaitCV.notify_all();
}
uint64_t Dispatcher::GetCompileBlockPtr() {
using ClassPtrType = void (FEXCore::Context::ContextImpl::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
PtrCast CompileBlockPtr;
CompileBlockPtr.ClassPtr = &FEXCore::Context::ContextImpl::CompileBlockJit;
return CompileBlockPtr.Data;
}
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx)
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
, CTX {ctx} {
, CTX {ctx}
, config {config} {
EmitDispatcher();
}
@@ -60,8 +85,8 @@ void Dispatcher::EmitDispatcher() {
// }
ARMEmitter::ForwardLabel l_CTX;
ARMEmitter::SingleUseForwardLabel l_Sleep;
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
ARMEmitter::ForwardLabel l_Sleep;
ARMEmitter::ForwardLabel l_CompileBlock;
// Push all the register we need to save
PushCalleeSavedRegisters();
@@ -78,7 +103,9 @@ void Dispatcher::EmitDispatcher() {
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
FillStaticRegs();
if (config.StaticRegisterAllocation) {
FillStaticRegs();
}
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
@@ -90,86 +117,87 @@ void Dispatcher::EmitDispatcher() {
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
// Load in our RIP
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
// Don't modify x2 since it contains our RIP once the block doesn't exist
auto RipReg = TMP3;
auto RipReg = ARMEmitter::XReg::x2;
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
// L1 Cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL , 4);
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP1, TMP1, 0);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
sub(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, RipReg);
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &FullLookup);
br(TMP4);
br(ARMEmitter::Reg::r3);
// L1C check failed, do a full lookup
Bind(&FullLookup);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), VirtualMemorySize - 1);
}
else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
}
ARMEmitter::ForwardLabel NoBlock;
{
// Offset the address and add to our page pointer
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
lsr(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 12);
// Load the pointer from the offset
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ExtendedType::LSL_64, 3);
// If page pointer is zero then we have no block
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 0x0FFF);
// Shift the offset by the size of the block cache entry
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
// The the full LookupCacheEntry with a single LDP.
// Check the guest address first to ensure it maps to the address we are currently at.
// This fixes aliasing problems
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
// If the guest address doesn't match, Compile the block.
sub(TMP2, TMP2, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
sub(ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, RipReg);
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, &NoBlock);
// Check the host address to see if it matches, else compile the block.
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
stp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP3, TMP1);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, 4);
stp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x2, ARMEmitter::Reg::r0);
// Jump to the block
br(TMP4);
br(ARMEmitter::Reg::r3);
}
}
{
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
@@ -182,7 +210,8 @@ void Dispatcher::EmitDispatcher() {
{
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
@@ -199,32 +228,26 @@ void Dispatcher::EmitDispatcher() {
blr(ARMEmitter::Reg::r2);
}
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
}
if (config.StaticRegisterAllocation)
FillStaticRegs();
FillStaticRegs();
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
// Trigger segfault if any deferred signals are pending
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
str(ARMEmitter::XReg::zr, TMP2, 0);
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
str(ARMEmitter::XReg::zr, ARMEmitter::XReg::x1, 0);
br(TMP1);
br(ARMEmitter::Reg::r0);
}
// Need to create the block
{
Bind(&NoBlock);
SpillStaticRegs(TMP1);
if (!TMP_ABIARGS) {
mov(ARMEmitter::XReg::x2, TMP3);
}
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
@@ -232,22 +255,22 @@ void Dispatcher::EmitDispatcher() {
ldr(ARMEmitter::XReg::x0, &l_CTX);
mov(ARMEmitter::XReg::x1, STATE);
// x2 contains guest RIP
mov(ARMEmitter::XReg::x3, 0);
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
ldr(ARMEmitter::XReg::x3, &l_CompileBlock);
// X2 contains our guest RIP
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uintptr_t, void *, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
GenerateIndirectRuntimeCall<void, void *, uint64_t, void *>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
}
FillStaticRegs();
if (config.StaticRegisterAllocation)
FillStaticRegs();
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
// Trigger segfault if any deferred signals are pending
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
@@ -277,7 +300,8 @@ void Dispatcher::EmitDispatcher() {
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
hlt(0);
}
@@ -287,7 +311,8 @@ void Dispatcher::EmitDispatcher() {
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
brk(0);
}
@@ -297,7 +322,8 @@ void Dispatcher::EmitDispatcher() {
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
// hlt/udf = SIGILL
// brk = SIGTRAP
@@ -317,7 +343,8 @@ void Dispatcher::EmitDispatcher() {
{
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
// We are pausing, this means the frontend should be waiting for this thread to idle
@@ -384,59 +411,112 @@ void Dispatcher::EmitDispatcher() {
str(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, State.rip));
// load static regs
FillStaticRegs();
if (config.StaticRegisterAllocation)
FillStaticRegs();
// Now go back to the regular dispatcher loop
b(&LoopTop);
}
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
auto Address = GetCursorAddress<uint64_t>();
{
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
if (!TMP_ABIARGS) {
mov(ARMEmitter::XReg::x0, TMP1);
mov(ARMEmitter::XReg::x1, TMP2);
mov(ARMEmitter::XReg::x2, TMP3);
}
ldr(ARMEmitter::XReg::x3, R, Offset);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
// Result is now in x0
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
return Address;
};
}
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
LUREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
LREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
{
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
Bind(&l_CTX);
dc64(reinterpret_cast<uintptr_t>(CTX));
Bind(&l_Sleep);
dc64(reinterpret_cast<uint64_t>(SleepThread));
Bind(&l_CompileBlock);
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::Context::ContextImpl::CompileBlock);
dc64(PMF.GetConvertedPointer());
dc64(GetCompileBlockPtr());
Start = reinterpret_cast<uint64_t>(DispatchPtr);
End = GetCursorAddress<uint64_t>();
@@ -455,7 +535,7 @@ void Dispatcher::EmitDispatcher() {
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
for (auto PCToDecode = DisasmBegin; PCToDecode < DisasmEnd; PCToDecode += 4) {
DisasmDecoder->Decode(PCToDecode);
auto Output = Disasm->GetOutput();
auto Output = Disasm.GetOutput();
LogMan::Msg::IFmt("{}", Output);
}
}
@@ -500,8 +580,8 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread)
}
}
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl *CTX) {
return fextl::make_unique<Dispatcher>(CTX);
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
return fextl::make_unique<Dispatcher>(CTX, Config);
}
}
@@ -2,8 +2,8 @@
#pragma once
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/fextl/memory.h>
#ifdef VIXL_SIMULATOR
@@ -31,14 +31,18 @@ class ContextImpl;
namespace FEXCore::CPU {
struct DispatcherConfig {
bool StaticRegisterAllocation = false;
};
#define STATE_PTR(STATE_TYPE, FIELD) \
STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
class Dispatcher final : public Arm64Emitter {
public:
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl *CTX);
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
Dispatcher(FEXCore::Context::ContextImpl *ctx);
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config);
~Dispatcher();
/**
@@ -102,8 +106,15 @@ public:
}
}
const DispatcherConfig& GetConfig() const { return config; }
protected:
FEXCore::Context::ContextImpl *CTX;
DispatcherConfig config;
static void SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame);
static uint64_t GetCompileBlockPtr();
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
+7 -6
View File
@@ -20,8 +20,8 @@ $end_info$
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/set.h>
#include <FEXHeaderUtils/TypeDefines.h>
namespace FEXCore::Frontend {
#include "Interface/Core/VSyscall/VSyscall.inc"
@@ -284,6 +284,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
if (DisplacementSize == 1) {
Literal = static_cast<int8_t>(Literal);
}
Displacement = DisplacementSize;
Operand->Type = DecodedOperand::OpType::GPRIndirect;
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
@@ -1126,11 +1127,11 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
// Entry is a jump target
BlocksToDecode.emplace(PC);
uint64_t CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
uint64_t CurrentCodePage = PC & FHU::FEX_PAGE_MASK;
fextl::set<uint64_t> CodePages = { CurrentCodePage };
AddContainedCodePage(PC, CurrentCodePage, FEXCore::Utils::FEX_PAGE_SIZE);
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
if (MaxInst == 0) {
MaxInst = CTX->Config.MaxInstPerBlock;
@@ -1156,8 +1157,8 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
auto OpMinAddress = RIPToDecode + PCOffset;
auto OpMaxAddress = OpMinAddress + MAX_INST_SIZE;
auto OpMinPage = OpMinAddress & FEXCore::Utils::FEX_PAGE_MASK;
auto OpMaxPage = OpMaxAddress & FEXCore::Utils::FEX_PAGE_MASK;
auto OpMinPage = OpMinAddress & FHU::FEX_PAGE_MASK;
auto OpMaxPage = OpMaxAddress & FHU::FEX_PAGE_MASK;
if (OpMinPage != CurrentCodePage) {
CurrentCodePage = OpMinPage;
@@ -1230,7 +1231,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
}
for (auto CodePage : CodePages) {
AddContainedCodePage(PC, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
AddContainedCodePage(PC, CodePage, FHU::FEX_PAGE_SIZE);
}
// sort for better branching
+115 -30
View File
@@ -9,6 +9,14 @@
#ifdef _M_X86_64
#define XBYAK64
#define XBYAK_CUSTOM_ALLOC
#define XBYAK_CUSTOM_MALLOC FEXCore::Allocator::malloc
#define XBYAK_CUSTOM_FREE FEXCore::Allocator::free
#define XBYAK_CUSTOM_SETS
#define XBYAK_STD_UNORDERED_SET fextl::unordered_set
#define XBYAK_STD_UNORDERED_MAP fextl::unordered_map
#define XBYAK_STD_UNORDERED_MULTIMAP fextl::unordered_multimap
#define XBYAK_STD_LIST fextl::list
#define XBYAK_NO_EXCEPTION
#include <FEXCore/fextl/list.h>
#include <FEXCore/fextl/unordered_map.h>
@@ -61,42 +69,114 @@ static void OverrideFeatures(HostFeatures *Features) {
return;
}
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
do { \
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
const bool AlreadyEnabled = Features->FeatureName; \
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
Features->FeatureName = Result; \
} while (0)
#define GET_SINGLE_OPTION(name, enum_name) \
#define ENABLE_DISABLE_OPTION(name, enum_name) \
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
ENABLE_DISABLE_OPTION(SupportsAVX, AVX, AVX);
ENABLE_DISABLE_OPTION(SupportsAVX2, AVX2, AVX2);
ENABLE_DISABLE_OPTION(SupportsSVE, SVE, SVE);
ENABLE_DISABLE_OPTION(SupportsAFP, AFP, AFP);
ENABLE_DISABLE_OPTION(SupportsRCPC, LRCPC, LRCPC);
ENABLE_DISABLE_OPTION(SupportsTSOImm9, LRCPC2, LRCPC2);
ENABLE_DISABLE_OPTION(SupportsCSSC, CSSC, CSSC);
ENABLE_DISABLE_OPTION(SupportsPMULL_128Bit, PMULL128, PMULL128);
ENABLE_DISABLE_OPTION(SupportsRAND, RNG, RNG);
ENABLE_DISABLE_OPTION(SupportsCLZERO, CLZERO, CLZERO);
ENABLE_DISABLE_OPTION(SupportsAtomics, Atomics, ATOMICS);
ENABLE_DISABLE_OPTION(SupportsFCMA, FCMA, FCMA);
ENABLE_DISABLE_OPTION(SupportsFlagM, FlagM, FLAGM);
ENABLE_DISABLE_OPTION(SupportsFlagM2, FlagM2, FLAGM2);
ENABLE_DISABLE_OPTION(SupportsRPRES, RPRES, RPRES);
ENABLE_DISABLE_OPTION(SupportsPreserveAllABI, PRESERVEALLABI, PRESERVEALLABI);
GET_SINGLE_OPTION(Crypto, CRYPTO);
ENABLE_DISABLE_OPTION(AVX, AVX);
ENABLE_DISABLE_OPTION(AVX2, AVX2);
ENABLE_DISABLE_OPTION(SVE, SVE);
ENABLE_DISABLE_OPTION(AFP, AFP);
ENABLE_DISABLE_OPTION(LRCPC, LRCPC);
ENABLE_DISABLE_OPTION(LRCPC2, LRCPC2);
ENABLE_DISABLE_OPTION(CSSC, CSSC);
ENABLE_DISABLE_OPTION(PMULL128, PMULL128);
ENABLE_DISABLE_OPTION(RNG, RNG);
ENABLE_DISABLE_OPTION(CLZERO, CLZERO);
ENABLE_DISABLE_OPTION(Atomics, ATOMICS);
ENABLE_DISABLE_OPTION(FCMA, FCMA);
ENABLE_DISABLE_OPTION(FlagM, FLAGM);
ENABLE_DISABLE_OPTION(FlagM2, FLAGM2);
ENABLE_DISABLE_OPTION(Crypto, CRYPTO);
ENABLE_DISABLE_OPTION(RPRES, RPRES);
#undef ENABLE_DISABLE_OPTION
#undef GET_SINGLE_OPTION
if (EnableAVX) {
Features->SupportsAVX = true;
}
else if (DisableAVX) {
Features->SupportsAVX = false;
}
if (EnableAVX2) {
Features->SupportsAVX2 = true;
}
else if (DisableAVX2) {
Features->SupportsAVX2 = false;
}
if (EnableSVE) {
Features->SupportsSVE = true;
}
else if (DisableSVE) {
Features->SupportsSVE = false;
}
if (EnableAFP) {
Features->SupportsAFP = true;
}
else if (DisableAFP) {
Features->SupportsAFP = false;
}
if (EnableLRCPC) {
Features->SupportsRCPC = true;
}
else if (DisableLRCPC) {
Features->SupportsRCPC = false;
}
if (EnableLRCPC2) {
Features->SupportsTSOImm9 = true;
}
else if (DisableLRCPC2) {
Features->SupportsTSOImm9 = false;
}
if (EnableCSSC) {
Features->SupportsCSSC = true;
}
else if (DisableCSSC) {
Features->SupportsCSSC = false;
}
if (EnablePMULL128) {
Features->SupportsPMULL_128Bit = true;
}
else if (DisablePMULL128) {
Features->SupportsPMULL_128Bit = false;
}
if (EnableRNG) {
Features->SupportsRAND = true;
}
else if (DisableRNG) {
Features->SupportsRAND = false;
}
if (EnableCLZERO) {
Features->SupportsCLZERO = true;
}
else if (DisableCLZERO) {
Features->SupportsCLZERO = false;
}
if (EnableAtomics) {
Features->SupportsAtomics = true;
}
else if (DisableAtomics) {
Features->SupportsAtomics = false;
}
if (EnableFCMA) {
Features->SupportsFCMA = true;
}
else if (DisableFCMA) {
Features->SupportsFCMA = false;
}
if (EnableFlagM) {
Features->SupportsFlagM = true;
}
else if (DisableFlagM) {
Features->SupportsFlagM = false;
}
if (EnableFlagM2) {
Features->SupportsFlagM2 = true;
}
else if (DisableFlagM2) {
Features->SupportsFlagM2 = false;
}
if (EnableCrypto) {
Features->SupportsAES = true;
Features->SupportsCRC = true;
@@ -109,6 +189,12 @@ static void OverrideFeatures(HostFeatures *Features) {
Features->SupportsSHA = false;
Features->SupportsPMULL_128Bit = false;
}
if (EnableRPRES) {
Features->SupportsRPRES = true;
}
else if (DisableRPRES) {
Features->SupportsRPRES = false;
}
}
HostFeatures::HostFeatures() {
@@ -253,7 +339,6 @@ HostFeatures::HostFeatures() {
SupportsFloatExceptions = true;
#endif
#endif
SupportsPreserveAllABI = FEXCORE_HAS_PRESERVE_ALL_ATTR;
OverrideFeatures(this);
}
}
@@ -0,0 +1,2 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/Debug/InternalThreadState.h>
@@ -2,8 +2,9 @@
#include "Common/SoftFloat.h"
#include "Common/SoftFloat-3e/softfloat.h"
#include <FEXCore/IR/IR.h>
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
#include "Interface/IR/IR.h"
namespace FEXCore::CPU {
FEXCORE_PRESERVE_ALL_ATTR
@@ -85,7 +85,7 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
Info[Core::OPINDEX_VPCMPISTRX] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle);
}
bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_Header const *IROp, FallbackInfo *Info) {
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
uint8_t OpSize = IROp->Size;
switch(IROp->Op) {
case IR::OP_F80CVTTO: {
@@ -93,11 +93,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_He
switch (Op->SrcSize) {
case 4: {
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
case 8: {
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
@@ -107,11 +107,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_He
case IR::OP_F80CVT: {
switch (OpSize) {
case 4: {
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
case 8: {
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
@@ -124,28 +124,28 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_He
switch (OpSize) {
case 2: {
if (Op->Truncate) {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2, SupportsPreserveAllABI};
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
else {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, SupportsPreserveAllABI};
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
return true;
}
case 4: {
if (Op->Truncate) {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4, SupportsPreserveAllABI};
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
else {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, SupportsPreserveAllABI};
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
return true;
}
case 8: {
if (Op->Truncate) {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8, SupportsPreserveAllABI};
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
else {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, SupportsPreserveAllABI};
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
return true;
}
@@ -167,7 +167,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_He
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
};
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags), SupportsPreserveAllABI};
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags), FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
@@ -176,11 +176,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_He
switch (Op->SrcSize) {
case 2: {
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
case 4: {
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
@@ -190,13 +190,13 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_He
#define COMMON_UNARY_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, FEXCORE_HAS_PRESERVE_ALL_ATTR}; \
return true; \
}
#define COMMON_BINARY_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, FEXCORE_HAS_PRESERVE_ALL_ATTR}; \
return true; \
}
@@ -244,10 +244,10 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_He
// SSE4.2 Fallbacks
case IR::OP_VPCMPESTRX:
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX, SupportsPreserveAllABI};
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
case IR::OP_VPCMPISTRX:
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
default:
@@ -7,6 +7,7 @@
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::IR {
class IRListView;
@@ -44,6 +45,6 @@ namespace FEXCore::CPU {
class InterpreterOps {
public:
static void FillFallbackIndexPointers(uint64_t *Info);
static bool GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_Header const *IROp, FallbackInfo *Info);
static bool GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info);
};
} // namespace FEXCore::CPU
@@ -5,7 +5,6 @@ tags: backend|arm64
$end_info$
*/
#include "FEXCore/IR/IR.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
@@ -86,51 +85,18 @@ DEF_OP(Add) {
}
}
DEF_OP(AddWithFlags) {
auto Op = IROp->C<IR::IROp_AddWithFlags>();
const uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
uint64_t Const;
if (IsInlineConstant(Op->Src2, &Const)) {
adds(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), Const);
} else {
adds(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
}
DEF_OP(AddShift) {
auto Op = IROp->C<IR::IROp_AddShift>();
const uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
add(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
DEF_OP(AddNZCV) {
auto Op = IROp->C<IR::IROp_AddNZCV>();
const uint8_t OpSize = IROp->Size;
const auto OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
auto Src1 = GetReg(Op->Src1.ID());
uint64_t Const;
if (IsInlineConstant(Op->Src2, &Const)) {
LOGMAN_THROW_AA_FMT(OpSize >= 4, "Constant not allowed here");
cmn(EmitSize, Src1, Const);
cmn(EmitSize, GetReg(Op->Src1.ID()), Const);
} else {
unsigned Shift = OpSize < 4 ? (32 - (8 * OpSize)) : 0;
if (OpSize < 4) {
lsl(ARMEmitter::Size::i32Bit, TMP1, Src1, Shift);
cmn(EmitSize, TMP1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
} else {
cmn(EmitSize, Src1, GetReg(Op->Src2.ID()));
}
cmn(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
}
@@ -144,36 +110,6 @@ DEF_OP(AdcNZCV) {
adcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
DEF_OP(AdcWithFlags) {
auto Op = IROp->C<IR::IROp_AdcWithFlags>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
adcs(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
}
DEF_OP(Adc) {
auto Op = IROp->C<IR::IROp_Adc>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
adc(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
}
DEF_OP(SbbWithFlags) {
auto Op = IROp->C<IR::IROp_SbbWithFlags>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
sbcs(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
DEF_OP(SbbNZCV) {
auto Op = IROp->C<IR::IROp_SbbNZCV>();
const auto OpSize = IROp->Size;
@@ -184,16 +120,6 @@ DEF_OP(SbbNZCV) {
sbcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
DEF_OP(Sbb) {
auto Op = IROp->C<IR::IROp_Sbb>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
sbc(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
}
DEF_OP(TestNZ) {
auto Op = IROp->C<IR::IROp_TestNZ>();
const uint8_t OpSize = IROp->Size;
@@ -242,7 +168,7 @@ DEF_OP(Sub) {
if (IsInlineConstant(Op->Src2, &Const)) {
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), Const);
} else {
sub(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
}
@@ -256,72 +182,21 @@ DEF_OP(SubShift) {
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
DEF_OP(SubWithFlags) {
auto Op = IROp->C<IR::IROp_SubWithFlags>();
const uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
uint64_t Const;
if (IsInlineConstant(Op->Src2, &Const)) {
subs(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), Const);
} else {
subs(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
}
}
DEF_OP(SubNZCV) {
auto Op = IROp->C<IR::IROp_SubNZCV>();
const uint8_t OpSize = IROp->Size;
const auto OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
uint64_t Const;
if (IsInlineConstant(Op->Src2, &Const)) {
LOGMAN_THROW_AA_FMT(OpSize >= 4, "Constant not allowed here");
cmp(EmitSize, GetReg(Op->Src1.ID()), Const);
} else if (IsInlineConstant(Op->Src1, &Const)) {
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
cmp(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src2.ID()));
} else {
unsigned Shift = OpSize < 4 ? (32 - (8 * OpSize)) : 0;
ARMEmitter::Register ShiftedSrc1 = GetZeroableReg(Op->Src1);
// Shift to fix flags for <32-bit ops.
// Any shift of zero is still zero so optimize out silly zero shifts.
if (OpSize < 4 && ShiftedSrc1 != ARMEmitter::Reg::zr) {
lsl(ARMEmitter::Size::i32Bit, TMP1, ShiftedSrc1, Shift);
ShiftedSrc1 = TMP1;
}
if (OpSize < 4) {
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
} else {
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()));
}
}
}
DEF_OP(CmpPairZ) {
auto Op = IROp->C<IR::IROp_CmpPairZ>();
const uint8_t OpSize = IROp->Size;
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
// Save NZCV
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
// Compare, setting Z and clobbering NzCV
const auto Src1 = GetRegPair(Op->Src1.ID());
const auto Src2 = GetRegPair(Op->Src2.ID());
cmp(EmitSize, Src1.first, Src2.first);
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
// Restore NzCV
if (CTX->HostFeatures.SupportsFlagM) {
rmif(TMP1, 0, 0xb /* NzCV */);
} else {
cset(ARMEmitter::Size::i32Bit, TMP2, ARMEmitter::Condition::CC_EQ);
bfi(ARMEmitter::Size::i32Bit, TMP1, TMP2, 30 /* lsb: Z */, 1);
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
cmp(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
}
@@ -334,21 +209,7 @@ DEF_OP(RmifNZCV) {
auto Op = IROp->C<IR::IROp_RmifNZCV>();
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
rmif(GetZeroableReg(Op->Src).X(), Op->Rotate, Op->Mask);
}
DEF_OP(SetSmallNZV) {
auto Op = IROp->C<IR::IROp_SetSmallNZV>();
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
const uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 1 || OpSize == 2, "Unsupported {} size: {}", __func__, OpSize);
if (OpSize == 1) {
setf8(GetReg(Op->Src.ID()).W());
} else {
setf16(GetReg(Op->Src.ID()).W());
}
rmif(GetReg(Op->Src.ID()).X(), Op->Rotate, Op->Mask);
}
DEF_OP(AXFlag) {
@@ -393,7 +254,9 @@ DEF_OP(CondAddNZCV) {
ARMEmitter::StatusFlags Flags = (ARMEmitter::StatusFlags)Op->FalseNZCV;
uint64_t Const = 0;
auto Src1 = GetZeroableReg(Op->Src1);
auto Src1 = IsInlineConstant(Op->Src1, &Const) ? ARMEmitter::Reg::zr :
GetReg(Op->Src1.ID());
LOGMAN_THROW_A_FMT(Const == 0, "Unsupported inline constant");
if (IsInlineConstant(Op->Src2, &Const)) {
ccmn(EmitSize, Src1, Const, Flags, MapSelectCC(Op->Cond));
@@ -402,24 +265,6 @@ DEF_OP(CondAddNZCV) {
}
}
DEF_OP(CondSubNZCV) {
auto Op = IROp->C<IR::IROp_CondSubNZCV>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
ARMEmitter::StatusFlags Flags = (ARMEmitter::StatusFlags)Op->FalseNZCV;
uint64_t Const = 0;
auto Src1 = GetZeroableReg(Op->Src1);
if (IsInlineConstant(Op->Src2, &Const)) {
ccmp(EmitSize, Src1, Const, Flags, MapSelectCC(Op->Cond));
} else {
ccmp(EmitSize, Src1, GetReg(Op->Src2.ID()), Flags, MapSelectCC(Op->Cond));
}
}
DEF_OP(Neg) {
auto Op = IROp->C<IR::IROp_Neg>();
const uint8_t OpSize = IROp->Size;
@@ -453,16 +298,6 @@ DEF_OP(UMul) {
mul(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
}
DEF_OP(UMull) {
auto Op = IROp->C<IR::IROp_UMull>();
umull(GetReg(Node).X(), GetReg(Op->Src1.ID()).W(), GetReg(Op->Src2.ID()).W());
}
DEF_OP(SMull) {
auto Op = IROp->C<IR::IROp_SMull>();
smull(GetReg(Node).X(), GetReg(Op->Src1.ID()).W(), GetReg(Op->Src2.ID()).W());
}
DEF_OP(Div) {
auto Op = IROp->C<IR::IROp_Div>();
@@ -708,41 +543,6 @@ DEF_OP(And) {
}
}
DEF_OP(AndWithFlags) {
auto Op = IROp->C<IR::IROp_AndWithFlags>();
const uint8_t OpSize = IROp->Size;
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
uint64_t Const;
const auto Dst = GetReg(Node);
auto Src1 = GetReg(Op->Src1.ID());
// See TestNZ
if (OpSize < 4) {
if (IsInlineConstant(Op->Src2, &Const)) {
and_(EmitSize, Dst, Src1, Const);
} else {
auto Src2 = GetReg(Op->Src2.ID());
if (Src1 != Src2) {
and_(EmitSize, Dst, Src1, Src2);
} else if (Dst != Src1) {
mov(ARMEmitter::Size::i64Bit, Dst, Src1);
}
}
unsigned Shift = 32 - (OpSize * 8);
cmn(EmitSize, ARMEmitter::Reg::zr, Dst, ARMEmitter::ShiftType::LSL, Shift);
} else {
if (IsInlineConstant(Op->Src2, &Const)) {
ands(EmitSize, Dst, Src1, Const);
} else {
const auto Src2 = GetReg(Op->Src2.ID());
ands(EmitSize, Dst, Src1, Src2);
}
}
}
DEF_OP(Andn) {
auto Op = IROp->C<IR::IROp_Andn>();
const uint8_t OpSize = IROp->Size;
@@ -787,16 +587,6 @@ DEF_OP(XorShift) {
eor(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
DEF_OP(XornShift) {
auto Op = IROp->C<IR::IROp_XornShift>();
const uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
eon(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
DEF_OP(Lshl) {
auto Op = IROp->C<IR::IROp_Lshl>();
const uint8_t OpSize = IROp->Size;
@@ -903,58 +693,64 @@ DEF_OP(PDep) {
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto Input = GetReg(Op->Input.ID());
const auto Mask = GetReg(Op->Mask.ID());
const auto Dest = GetReg(Node);
// PDep implementation follows the ideas from
// http://0x80.pl/articles/pdep-soft-emu.html ... Basically, iterate the *set*
// bits only, which will be faster than the naive implementation as long as
// there are enough holes in the mask.
//
// The specific arm64 assembly used is based on the sequence that clang
// generates for the C code, giving context to the scheduling yielding better
// ILP than I would do by hand. The registers are allocated by hand however,
// to fit within the tight constraints we have here withot spilling. Also, we
// use cbz/cbnz for conditional branching to avoid clobbering NZCV.
const auto ShiftedBitReg = TMP1.R();
const auto BitReg = TMP2.R();
const auto SubMaskReg = TMP3.R();
const auto IndexReg = TMP4.R();
const auto ZeroReg = ARMEmitter::Reg::zr;
// We can't clobber these
const auto OrigInput = GetReg(Op->Input.ID());
const auto OrigMask = GetReg(Op->Mask.ID());
const auto InputReg = StaticRegisters[0];
const auto MaskReg = StaticRegisters[1];
const auto DestReg = StaticRegisters[2];
// So we have shadow as temporaries
const auto Input = TMP1.R();
const auto Mask = TMP2.R();
// these get used variously as scratch
const auto T0 = TMP3.R();
const auto T1 = TMP4.R();
const auto SpillCode = 1U << InputReg.Idx() |
1U << MaskReg.Idx() |
1U << DestReg.Idx();
ARMEmitter::ForwardLabel EarlyExit;
ARMEmitter::BackwardLabel NextBit;
ARMEmitter::SingleUseForwardLabel Done;
ARMEmitter::ForwardLabel Done;
cbz(EmitSize, Mask, &EarlyExit);
mov(EmitSize, IndexReg, ZeroReg);
// First, copy the input/mask, since we'll be clobbering. Copy as 64-bit to
// make this 0-uop on Firestorm.
mov(ARMEmitter::Size::i64Bit, Input, OrigInput);
mov(ARMEmitter::Size::i64Bit, Mask, OrigMask);
// We sadly need to spill regs for this for the time being
// TODO: Remove when scratch registers can be allocated
// explicitly.
SpillStaticRegs(TMP1, false, SpillCode);
// Now, they're copied, so we can start setting Dest (even if it overlaps with
// one of them). Handle early exit case
mov(EmitSize, Dest, 0);
cbz(EmitSize, OrigMask, &Done);
// Setup for first iteration
neg(EmitSize, T0, Mask);
and_(EmitSize, T0, T0, Mask);
mov(EmitSize, InputReg, Input);
mov(EmitSize, MaskReg, Mask);
mov(EmitSize, DestReg, ZeroReg);
// Main loop
Bind(&NextBit);
sbfx(EmitSize, T1, Input, 0, 1);
eor(EmitSize, Mask, Mask, T0);
and_(EmitSize, T0, T1, T0);
neg(EmitSize, T1, Mask);
orr(EmitSize, Dest, Dest, T0);
lsr(EmitSize, Input, Input, 1);
and_(EmitSize, T0, Mask, T1);
cbnz(EmitSize, T0, &NextBit);
rbit(EmitSize, ShiftedBitReg, MaskReg);
clz(EmitSize, ShiftedBitReg, ShiftedBitReg);
lsrv(EmitSize, BitReg, InputReg, IndexReg);
and_(EmitSize, BitReg, BitReg, 1);
sub(EmitSize, SubMaskReg, MaskReg, 1);
add(EmitSize, IndexReg, IndexReg, 1);
ands(EmitSize, MaskReg, MaskReg, SubMaskReg);
lslv(EmitSize, ShiftedBitReg, BitReg, ShiftedBitReg);
orr(EmitSize, DestReg, DestReg, ShiftedBitReg);
b(ARMEmitter::Condition::CC_NE, &NextBit);
// Store result in a temp so it doesn't get clobbered.
// and restore it after the re-fill below.
mov(EmitSize, IndexReg, DestReg);
// Restore our registers before leaving
// TODO: Also remove along with above TODO.
FillStaticRegs(false, SpillCode);
mov(EmitSize, Dest, IndexReg);
b(&Done);
// Early exit
Bind(&EarlyExit);
mov(EmitSize, Dest, ZeroReg);
// All done with nothing to do.
Bind(&Done);
@@ -976,9 +772,9 @@ DEF_OP(PExt) {
const auto BitReg = TMP2;
const auto ValueReg = TMP3;
ARMEmitter::SingleUseForwardLabel EarlyExit;
ARMEmitter::ForwardLabel EarlyExit;
ARMEmitter::BackwardLabel NextBit;
ARMEmitter::SingleUseForwardLabel Done;
ARMEmitter::ForwardLabel Done;
cbz(EmitSize, Mask, &EarlyExit);
mov(EmitSize, MaskReg, Mask);
@@ -1032,8 +828,8 @@ DEF_OP(LDiv) {
break;
}
case 8: {
ARMEmitter::SingleUseForwardLabel Only64Bit{};
ARMEmitter::SingleUseForwardLabel LongDIVRet{};
ARMEmitter::ForwardLabel Only64Bit{};
ARMEmitter::ForwardLabel LongDIVRet{};
// Check if the upper bits match the top bit of the lower 64-bits
// Sign extend the top bit of lower bits
@@ -1045,18 +841,18 @@ DEF_OP(LDiv) {
// Long divide
{
mov(EmitSize, TMP1, Upper);
mov(EmitSize, TMP2, Lower);
mov(EmitSize, TMP3, Divisor);
mov(EmitSize, ARMEmitter::Reg::r0, Upper);
mov(EmitSize, ARMEmitter::Reg::r1, Lower);
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP4);
blr(ARMEmitter::Reg::r3);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
// Move result to its destination register
mov(EmitSize, Dst, TMP1);
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
// Skip 64-bit path
b(&LongDIVRet);
@@ -1104,8 +900,8 @@ DEF_OP(LUDiv) {
break;
}
case 8: {
ARMEmitter::SingleUseForwardLabel Only64Bit{};
ARMEmitter::SingleUseForwardLabel LongDIVRet{};
ARMEmitter::ForwardLabel Only64Bit{};
ARMEmitter::ForwardLabel LongDIVRet{};
// Check the upper bits for zero
// If the upper bits are zero then we can do a 64-bit divide
@@ -1113,18 +909,18 @@ DEF_OP(LUDiv) {
// Long divide
{
mov(EmitSize, TMP1, Upper);
mov(EmitSize, TMP2, Lower);
mov(EmitSize, TMP3, Divisor);
mov(EmitSize, ARMEmitter::Reg::r0, Upper);
mov(EmitSize, ARMEmitter::Reg::r1, Lower);
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP4);
blr(ARMEmitter::Reg::r3);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
// Move result to its destination register
mov(EmitSize, Dst, TMP1);
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
// Skip 64-bit path
b(&LongDIVRet);
@@ -1176,8 +972,8 @@ DEF_OP(LRem) {
break;
}
case 8: {
ARMEmitter::SingleUseForwardLabel Only64Bit{};
ARMEmitter::SingleUseForwardLabel LongDIVRet{};
ARMEmitter::ForwardLabel Only64Bit{};
ARMEmitter::ForwardLabel LongDIVRet{};
// Check if the upper bits match the top bit of the lower 64-bits
// Sign extend the top bit of lower bits
@@ -1189,18 +985,18 @@ DEF_OP(LRem) {
// Long divide
{
mov(EmitSize, TMP1, Upper);
mov(EmitSize, TMP2, Lower);
mov(EmitSize, TMP3, Divisor);
mov(EmitSize, ARMEmitter::Reg::r0, Upper);
mov(EmitSize, ARMEmitter::Reg::r1, Lower);
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREMHandler));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREMHandler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP4);
blr(ARMEmitter::Reg::r3);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
// Move result to its destination register
mov(EmitSize, Dst, TMP1);
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
// Skip 64-bit path
b(&LongDIVRet);
@@ -1250,8 +1046,8 @@ DEF_OP(LURem) {
break;
}
case 8: {
ARMEmitter::SingleUseForwardLabel Only64Bit{};
ARMEmitter::SingleUseForwardLabel LongDIVRet{};
ARMEmitter::ForwardLabel Only64Bit{};
ARMEmitter::ForwardLabel LongDIVRet{};
// Check the upper bits for zero
// If the upper bits are zero then we can do a 64-bit divide
@@ -1259,18 +1055,18 @@ DEF_OP(LURem) {
// Long divide
{
mov(EmitSize, TMP1, Upper);
mov(EmitSize, TMP2, Lower);
mov(EmitSize, TMP3, Divisor);
mov(EmitSize, ARMEmitter::Reg::r0, Upper);
mov(EmitSize, ARMEmitter::Reg::r1, Lower);
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREMHandler));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREMHandler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP4);
blr(ARMEmitter::Reg::r3);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
// Move result to its destination register
mov(EmitSize, Dst, TMP1);
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
// Skip 64-bit path
b(&LongDIVRet);
@@ -1403,11 +1199,6 @@ DEF_OP(FindTrailingZeroes) {
rbit(EmitSize, Dst, Src);
if (OpSize == 2) {
// This orr does two things. First, if the (masked) source is zero, it
// reverses to zero in the top so it forces clz to return 16. Second, it
// ensures garbage in the upper bits of the source don't affect clz, because
// they'll rbit to garbage in the bottom below the 0x8000 and be ignored by
// the clz. So we handle Src upper garbage without explicitly masking.
orr(EmitSize, Dst, Dst, 0x8000);
}
@@ -1425,8 +1216,6 @@ DEF_OP(CountLeadingZeroes) {
const auto Src = GetReg(Op->Src.ID());
if (OpSize == 2) {
// Expressing as lsl+orr+clz clears away any garbage in the upper bits
// (alternatively could do uxth+clz+sub.. equal cost in total).
lsl(EmitSize, Dst, Src, 16);
orr(EmitSize, Dst, Dst, 0x8000);
clz(EmitSize, Dst, Dst);
@@ -1625,8 +1414,11 @@ DEF_OP(NZCVSelect) {
csetm(EmitSize, Dst, cc);
else
cset(EmitSize, Dst, cc);
} else if (is_const_false) {
LOGMAN_THROW_A_FMT(const_false == 0, "NZCVSelect: unsupported constant");
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), ARMEmitter::Reg::zr, cc);
} else {
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), GetZeroableReg(Op->FalseVal), cc);
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), GetReg(Op->FalseVal.ID()), cc);
}
}
@@ -31,19 +31,11 @@ DEF_OP(CASPair) {
mov(EmitSize, Dst.second, TMP4.R());
}
else {
// Save NZCV so we don't have to mark this op as clobbering NZCV (the
// SupportsAtomics does not clobber atomics and this !SupportsAtomics path
// is so slow it's not worth the complexity of splitting the IR op.). We
// clobber NZCV inside the hot loop and we can't replace cmp/ccmp/b.ne with
// something NZCV-preserving without requiring an extra instruction.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
ARMEmitter::SingleUseForwardLabel LoopExpected;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
Bind(&LoopTop);
// This instruction sequence must be synced with HandleCASPAL_Armv8.
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
cmp(EmitSize, TMP2, Expected.first);
ccmp(EmitSize, TMP3, Expected.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
@@ -62,9 +54,6 @@ DEF_OP(CASPair) {
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopExpected);
// Restore
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
}
}
@@ -93,8 +82,8 @@ DEF_OP(CAS) {
}
else {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
ARMEmitter::SingleUseForwardLabel LoopExpected;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
if (OpSize == 1) {
@@ -321,7 +310,8 @@ DEF_OP(AtomicSwap) {
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
mov(EmitSize, TMP2, Src);
ldswpal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
}
else {
ARMEmitter::BackwardLabel LoopTop;
@@ -11,9 +11,9 @@ $end_info$
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "Interface/Core/InternalThreadState.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/MathUtils.h>
#include <Interface/HLE/Thunks/Thunks.h>
@@ -53,30 +53,33 @@ DEF_OP(ExitFunction) {
uint64_t NewRIP;
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
ARMEmitter::SingleUseForwardLabel l_BranchHost;
ARMEmitter::ForwardLabel l_BranchHost;
ARMEmitter::ForwardLabel l_BranchGuest;
ldr(TMP1, &l_BranchHost);
blr(TMP1);
ldr(ARMEmitter::XReg::x0, &l_BranchHost);
blr(ARMEmitter::Reg::r0);
Bind(&l_BranchHost);
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
Bind(&l_BranchGuest);
dc64(NewRIP);
} else {
ARMEmitter::SingleUseForwardLabel FullLookup;
ARMEmitter::ForwardLabel FullLookup;
auto RipReg = GetReg(Op->NewRIP.ID());
// L1 Cache
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg, LookupCache::L1_ENTRIES_MASK);
add(TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x3, ARMEmitter::ShiftType::LSL, 4);
// Note: sub+cbnz used over cmp+br to preserve flags.
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
sub(TMP1, TMP1, RipReg.X());
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x1, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
sub(TMP1, ARMEmitter::XReg::x0, RipReg.X());
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
br(TMP2);
br(ARMEmitter::Reg::r1);
Bind(&FullLookup);
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
@@ -128,8 +131,8 @@ DEF_OP(CondJump) {
if (Op->FromNZCV) {
b(MapBranchCC(Op->Cond), TrueTargetLabel);
} else {
[[maybe_unused]] uint64_t Const;
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
@@ -350,58 +353,58 @@ DEF_OP(ValidateCode) {
int idx = 0;
LoadConstant(ARMEmitter::Size::i64Bit, GetReg(Node), 0);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Entry + Op->Offset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, 1);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Entry + Op->Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 1);
const auto Dst = GetReg(Node);
while (len >= 8)
{
ldr(ARMEmitter::XReg::x2, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i64Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 8;
idx += 8;
}
while (len >= 4)
{
ldr(ARMEmitter::WReg::w2, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
ldr(ARMEmitter::WReg::w2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 4;
idx += 4;
}
while (len >= 2)
{
ldrh(TMP3, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint16_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
ldrh(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint16_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 2;
idx += 2;
}
while (len >= 1)
{
ldrb(TMP3, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint8_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
ldrb(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint8_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 1;
idx += 1;
}
}
DEF_OP(ThreadRemoveCodeEntry) {
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
// Arguments are passed as follows:
// X0: Thread
// X1: RIP
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
PushDynamicRegsAndLR(TMP1);
SpillStaticRegs(TMP1);
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
@@ -420,23 +423,16 @@ DEF_OP(ThreadRemoveCodeEntry) {
DEF_OP(CPUID) {
auto Op = IROp->C<IR::IROp_CPUID>();
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function.ID()));
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf.ID()));
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
PushDynamicRegsAndLR(TMP1);
SpillStaticRegs(TMP1);
// x0 = CPUID Handler
// x1 = CPUID Function
// x2 = CPUID Leaf
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction));
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, TMP2);
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, TMP3);
}
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, GetReg(Op->Leaf.ID()));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<__uint128_t, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
@@ -444,11 +440,6 @@ DEF_OP(CPUID) {
blr(ARMEmitter::Reg::r3);
}
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
mov(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Reg::r1);
}
FillStaticRegs();
PopDynamicRegsAndLR();
@@ -456,22 +447,21 @@ DEF_OP(CPUID) {
// Results are in x0, x1
// Results want to be in a i64v2 vector
auto Dst = GetRegPair(Node);
mov(ARMEmitter::Size::i64Bit, Dst.first, TMP1);
mov(ARMEmitter::Size::i64Bit, Dst.second, TMP2);
mov(ARMEmitter::Size::i64Bit, Dst.first, ARMEmitter::Reg::r0);
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
}
DEF_OP(XGetBV) {
auto Op = IROp->C<IR::IROp_XGetBV>();
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
PushDynamicRegsAndLR(TMP1);
SpillStaticRegs(TMP1);
// x0 = CPUID Handler
// x1 = XCR Function
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
}
@@ -479,10 +469,6 @@ DEF_OP(XGetBV) {
blr(ARMEmitter::Reg::r2);
}
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
}
FillStaticRegs();
PopDynamicRegsAndLR();
@@ -490,8 +476,8 @@ DEF_OP(XGetBV) {
// Results are in x0
// Results want to be in a i32v2 vector
auto Dst = GetRegPair(Node);
mov(ARMEmitter::Size::i32Bit, Dst.first, TMP1);
lsr(ARMEmitter::Size::i64Bit, Dst.second, TMP1, 32);
mov(ARMEmitter::Size::i32Bit, Dst.first, ARMEmitter::Reg::r0);
lsr(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r0, 32);
}
#undef DEF_OP
@@ -201,7 +201,7 @@ DEF_OP(VSha256U0) {
else {
mov(VTMP1.Q(), Src1.Q());
sha256su0(VTMP1, Src2);
mov(Dst.Q(), VTMP1.Q());
mov(Dst.Q(), Src1.Q());
}
}
+109 -127
View File
@@ -18,18 +18,17 @@ $end_info$
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Utils/MemberFunctionToPointer.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include "Interface/Core/Interpreter/InterpreterOps.h"
@@ -79,45 +78,11 @@ namespace FEXCore::CPU {
void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
FallbackInfo Info;
if (!InterpreterOps::GetFallbackHandler(CTX->HostFeatures.SupportsPreserveAllABI, IROp, &Info)) {
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_MSG_A_FMT("Unhandled IR Op: {}", FEXCore::IR::GetName(IROp->Op));
#endif
} else {
auto FillF80Result = [&]() {
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
mov(TMP2, ARMEmitter::XReg::x1);
}
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
eor(Dst.Q(), Dst.Q(), Dst.Q());
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, TMP2);
};
auto FillF64Result = [&]() {
if (!TMP_ABIARGS) {
mov(VTMP1.D(), ARMEmitter::DReg::d0);
}
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
mov(Dst.D(), VTMP1.D());
};
auto FillI32Result = [&]() {
if (!TMP_ABIARGS) {
mov(TMP1.W(), ARMEmitter::WReg::w0);
}
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetReg(Node);
mov(Dst.W(), TMP1.W());
};
switch(Info.ABI) {
case FABI_F80_I16_F32:{
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
@@ -133,7 +98,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r1);
}
FillF80Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
eor(Dst.Q(), Dst.Q(), Dst.Q());
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
}
break;
@@ -151,7 +121,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r1);
}
FillF80Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
eor(Dst.Q(), Dst.Q(), Dst.Q());
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
}
break;
@@ -160,13 +135,13 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
const auto Src1 = GetReg(IROp->Args[0].ID());
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
if (Info.ABI == FABI_F80_I16_I16) {
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
}
else {
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
}
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint32_t>(ARMEmitter::Reg::r2);
@@ -175,7 +150,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r2);
}
FillF80Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
eor(Dst.Q(), Dst.Q(), Dst.Q());
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
}
break;
@@ -196,13 +176,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r3);
}
if (!TMP_ABIARGS) {
fmov(VTMP1.S(), ARMEmitter::SReg::s0);
}
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
fmov(Dst.S(), VTMP1.S());
fmov(Dst.S(), ARMEmitter::SReg::s0);
}
break;
@@ -223,7 +200,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r3);
}
FillF64Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
mov(Dst.D(), ARMEmitter::DReg::d0);
}
break;
@@ -242,24 +222,21 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r1);
}
FillF64Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
mov(Dst.D(), ARMEmitter::DReg::d0);
}
break;
case FABI_F64_I16_F64_F64: {
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
const auto Src1 = GetVReg(IROp->Args[0].ID());
const auto Src2 = GetVReg(IROp->Args[1].ID());
mov(VTMP1.D(), Src1.D());
mov(VTMP2.D(), Src2.D());
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
if (!TMP_ABIARGS) {
mov(ARMEmitter::DReg::d0, VTMP1.D());
mov(ARMEmitter::DReg::d1, VTMP2.D());
}
mov(ARMEmitter::DReg::d0, Src1.D());
mov(ARMEmitter::DReg::d1, Src2.D());
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
@@ -269,7 +246,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r1);
}
FillF64Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
mov(Dst.D(), ARMEmitter::DReg::d0);
}
break;
@@ -290,13 +270,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r3);
}
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
}
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetReg(Node);
sxth(ARMEmitter::Size::i64Bit, Dst, TMP1);
sxth(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
}
break;
case FABI_I32_I16_F80:{
@@ -316,7 +293,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r3);
}
FillI32Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetReg(Node);
mov(ARMEmitter::Size::i32Bit, Dst, ARMEmitter::Reg::r0);
}
break;
case FABI_I64_I16_F80:{
@@ -335,14 +315,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
else {
blr(ARMEmitter::Reg::r3);
}
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
}
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetReg(Node);
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
}
break;
case FABI_I64_I16_F80_F80:{
@@ -365,14 +341,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
else {
blr(ARMEmitter::Reg::r5);
}
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
}
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetReg(Node);
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
}
break;
case FABI_F80_I16_F80:{
@@ -392,7 +364,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r3);
}
FillF80Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
eor(Dst.Q(), Dst.Q(), Dst.Q());
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
}
break;
case FABI_F80_I16_F80_F80:{
@@ -416,28 +393,27 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r5);
}
FillF80Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetVReg(Node);
eor(Dst.Q(), Dst.Q(), Dst.Q());
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
}
break;
case FABI_I32_I64_I64_I128_I128_I16: {
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
const auto SrcRAX = GetReg(Op->RAX.ID());
const auto SrcRDX = GetReg(Op->RDX.ID());
mov(TMP1, SrcRAX.X());
mov(TMP2, SrcRDX.X());
SpillForABICall(Info.SupportsPreserveAllABI, TMP3, true);
const auto Control = Op->Control;
const auto Src1 = GetVReg(Op->LHS.ID());
const auto Src2 = GetVReg(Op->RHS.ID());
const auto SrcRAX = GetReg(Op->RAX.ID());
const auto SrcRDX = GetReg(Op->RDX.ID());
if (!TMP_ABIARGS) {
mov(ARMEmitter::XReg::x0, TMP1);
mov(ARMEmitter::XReg::x1, TMP2);
}
mov(ARMEmitter::XReg::x0, SrcRAX.X());
mov(ARMEmitter::XReg::x1, SrcRDX.X());
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src1, 0);
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src1, 1);
@@ -455,9 +431,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r7);
}
FillI32Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetReg(Node);
mov(Dst.W(), ARMEmitter::WReg::w0);
break;
}
break;
case FABI_I32_I128_I128_I16: {
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
@@ -483,9 +462,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
blr(ARMEmitter::Reg::r5);
}
FillI32Result();
FillForABICall(Info.SupportsPreserveAllABI, true);
const auto Dst = GetReg(Node);
mov(Dst.W(), ARMEmitter::WReg::w0);
break;
}
break;
case FABI_UNKNOWN:
default:
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
@@ -498,26 +480,9 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
}
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record) {
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
uintptr_t branch = (uintptr_t)(Record) - 8;
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 8);
FEXCore::ARMEmitter::SingleUseForwardLabel l_BranchHost;
emit.ldr(TMP1, &l_BranchHost);
emit.blr(TMP1);
emit.Bind(&l_BranchHost);
emit.dc64(LinkerAddress);
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 8);
}
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record) {
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
Record->HostBranch = LinkerAddress;
}
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record) {
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto GuestRip = Record->GuestRIP;
auto GuestRip = record[1];
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
@@ -526,24 +491,35 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
return Frame->Pointers.Common.DispatcherLoopTop;
}
uintptr_t branch = (uintptr_t)(Record) - 8;
uintptr_t branch = (uintptr_t)(record) - 8;
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
auto offset = HostCode/4 - branch/4;
if (vixl::IsInt26(offset)) {
// optimal case - can branch directly
// patch the code
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 4);
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
emit.b(offset);
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 4);
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
// Add de-linking handler
Thread->LookupCache->AddBlockLink(GuestRip, Record, DirectBlockDelinker);
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
FEXCore::ARMEmitter::ForwardLabel l_BranchHost;
emit.ldr(FEXCore::ARMEmitter::XReg::x0, &l_BranchHost);
emit.blr(FEXCore::ARMEmitter::Reg::r0);
emit.Bind(&l_BranchHost);
emit.dc64(LinkerAddress);
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
});
} else {
// fallback case - do a soft-er link by patching the pointer
Record->HostBranch = HostCode;
record[0] = HostCode;
// Add de-linking handler
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker);
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
record[0] = LinkerAddress;
});
}
return HostCode;
@@ -598,11 +574,8 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
Common.XCRFunction = PMF.GetConvertedPointer();
}
{
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::HLE::SyscallHandler::HandleSyscall);
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
Common.SyscallHandlerFunc = PMF.GetVTableEntry(CTX->SyscallHandler);
}
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
// Fill in the fallback handlers
@@ -621,6 +594,15 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
ClearCache();
// Setup dynamic dispatch.
if (CTX->Dispatcher->GetConfig().StaticRegisterAllocation) {
RT_LoadRegister = &Arm64JITCore::Op_LoadRegisterSRA;
RT_StoreRegister = &Arm64JITCore::Op_StoreRegisterSRA;
}
else {
RT_LoadRegister = &Arm64JITCore::Op_LoadRegister;
RT_StoreRegister = &Arm64JITCore::Op_StoreRegister;
}
if (ParanoidTSO()) {
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
@@ -780,8 +762,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
if (vixl::aarch64::Assembler::IsImmAddSub(TotalSpillSlotsSize)) {
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, TotalSpillSlotsSize);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, ARMEmitter::XReg::x0, ARMEmitter::ExtendedType::LSL_64, 0);
}
}
@@ -858,7 +840,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
// TODO: This needs to be a data RIP relocation once code caching works.
// Current relocation code doesn't support this feature yet.
JITBlockTail->RIP = Entry;
JITBlockTail->SpinLockFutex = 0;
{
// Store the RIP entries.
@@ -900,7 +881,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
LogMan::Msg::IFmt("Disassemble Begin");
for (auto PCToDecode = DisasmBegin; PCToDecode < DisasmEnd; PCToDecode += 4) {
DisasmDecoder->Decode(PCToDecode);
auto Output = Disasm->GetOutput();
auto Output = Disasm.GetOutput();
LogMan::Msg::IFmt("{}", Output);
}
LogMan::Msg::IFmt("Disassemble End");
@@ -928,8 +909,8 @@ void Arm64JITCore::ResetStack() {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
} else {
// Too big to fit in a 12bit immediate
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, TotalSpillSlotsSize);
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, ARMEmitter::XReg::x0, ARMEmitter::ExtendedType::LSL_64, 0);
}
}
@@ -939,6 +920,7 @@ fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *
CPUBackendFeatures GetArm64JITBackendFeatures() {
return CPUBackendFeatures {
.SupportsStaticRegisterAllocation = true,
.SupportsFlags = true,
.SupportsSaturatingRoundingShifts = true,
.SupportsVTBL2 = true,
@@ -9,17 +9,16 @@ $end_info$
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <aarch64/assembler-aarch64.h>
#include <aarch64/disasm-aarch64.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
@@ -57,8 +56,6 @@ public:
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
const bool HostSupportsSVE128{};
const bool HostSupportsSVE256{};
@@ -119,16 +116,6 @@ private:
return PhyReg;
}
[[nodiscard]] FEXCore::ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
uint64_t Const;
if (IsInlineConstant(Src, &Const)) {
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
return ARMEmitter::Reg::zr;
} else {
return GetReg(Src.ID());
}
}
// Converts IR-base shift type to ARMEmitter shift type.
// Will be a no-op, only a type conversion since the two definitions match.
[[nodiscard]] ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
@@ -239,6 +226,9 @@ private:
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
// Runtime selection;
// Load and store register style.
OpType RT_LoadRegister;
OpType RT_StoreRegister;
// Load and store TSO memory style
OpType RT_LoadMemTSO;
OpType RT_StoreMemTSO;
@@ -246,6 +236,9 @@ private:
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
// Dynamic Dispatcher supporting operations
DEF_OP(LoadRegisterSRA);
DEF_OP(StoreRegisterSRA);
DEF_OP(ParanoidLoadMemTSO);
DEF_OP(ParanoidStoreMemTSO);
@@ -131,6 +131,170 @@ DEF_OP(LoadRegister) {
const auto Op = IROp->C<IR::IROp_LoadRegister>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::GPRClass) {
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
const auto regOffs = Op->Offset & 7;
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
switch (OpSize) {
case 1:
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
ldrb(GetReg(Node), STATE, Op->Offset);
break;
case 2:
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
ldrh(GetReg(Node), STATE, Op->Offset);
break;
case 4:
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
ldr(GetReg(Node).W(), STATE, Op->Offset);
break;
case 8:
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
ldr(GetReg(Node).X(), STATE, Op->Offset);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadRegister GPR size: {}", OpSize);
break;
}
}
else if (Op->Class == IR::FPRClass) {
const auto regSize = HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE
: Core::CPUState::XMM_SSE_REG_SIZE;
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Unsupported code path!");
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
const auto host = GetVReg(Node);
const auto regOffs = Op->Offset & 15;
switch (OpSize) {
case 1: {
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
ldrb(host, STATE, Op->Offset);
break;
}
case 2: {
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
ldrh(host, STATE, Op->Offset);
break;
}
case 4: {
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
ldr(host.S(), STATE, Op->Offset);
break;
}
case 8: {
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
ldr(host.D(), STATE, Op->Offset);
break;
}
case 16: {
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
ldr(host.Q(), STATE, Op->Offset);
break;
}
}
} else {
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
}
}
DEF_OP(StoreRegister) {
const auto Op = IROp->C<IR::IROp_StoreRegister>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::GPRClass) {
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
const auto regOffs = Op->Offset & 7;
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
const auto Src = GetReg(Op->Value.ID());
switch (OpSize) {
case 1:
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
strb(Src, STATE, Op->Offset);
break;
case 2:
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
strh(Src, STATE, Op->Offset);
break;
case 4:
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
str(Src.W(), STATE, Op->Offset);
break;
case 8:
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
str(Src.X(), STATE, Op->Offset);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreRegister GPR size: {}", OpSize);
break;
}
} else if (Op->Class == IR::FPRClass) {
const auto regSize = HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE
: Core::CPUState::XMM_SSE_REG_SIZE;
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Unsupported code path!");
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
const auto host = GetVReg(Op->Value.ID());
const auto regOffs = Op->Offset & 15;
switch (OpSize) {
case 1:
strb(host, STATE, Op->Offset);
break;
case 2:
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs: {}", regOffs);
strh(host, STATE, Op->Offset);
break;
case 4:
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
str(host.S(), STATE, Op->Offset);
break;
case 8:
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
str(host.D(), STATE, Op->Offset);
break;
case 16:
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
str(host.Q(), STATE, Op->Offset);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreRegister FPR size: {}", OpSize);
break;
}
} else {
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
}
}
DEF_OP(LoadRegisterSRA) {
const auto Op = IROp->C<IR::IROp_LoadRegister>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::GPRClass) {
const auto regId =
Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ? (StaticRegisters.size() - 2) :
@@ -175,7 +339,7 @@ DEF_OP(LoadRegister) {
if (HostSupportsSVE256) {
const auto regOffs = Op->Offset & 31;
ARMEmitter::SingleUseForwardLabel DataLocation;
ARMEmitter::ForwardLabel DataLocation;
const auto LoadPredicate = [this, &DataLocation] {
const auto Predicate = ARMEmitter::PReg::p0;
adr(TMP1, &DataLocation);
@@ -184,7 +348,7 @@ DEF_OP(LoadRegister) {
};
const auto EmitData = [this, &DataLocation](uint32_t Value) {
ARMEmitter::SingleUseForwardLabel PastConstant;
ARMEmitter::ForwardLabel PastConstant;
b(&PastConstant);
Bind(&DataLocation);
dc32(Value);
@@ -309,7 +473,7 @@ DEF_OP(LoadRegister) {
}
}
DEF_OP(StoreRegister) {
DEF_OP(StoreRegisterSRA) {
const auto Op = IROp->C<IR::IROp_StoreRegister>();
const auto OpSize = IROp->Size;
@@ -364,7 +528,7 @@ DEF_OP(StoreRegister) {
const auto regOffs = Op->Offset & 31;
// Compartmentalized setting up of the predicate for the cases that need it.
ARMEmitter::SingleUseForwardLabel DataLocation;
ARMEmitter::ForwardLabel DataLocation;
const auto LoadPredicate = [this, &DataLocation] {
const auto Predicate = ARMEmitter::PReg::p0;
adr(TMP1, &DataLocation);
@@ -377,7 +541,7 @@ DEF_OP(StoreRegister) {
// It's helpful to treat LoadPredicate and EmitData as a prologue and epilogue
// respectfully.
const auto EmitData = [this, &DataLocation](uint32_t Data) {
ARMEmitter::SingleUseForwardLabel PastConstant;
ARMEmitter::ForwardLabel PastConstant;
b(&PastConstant);
Bind(&DataLocation);
dc32(Data);
@@ -888,14 +1052,6 @@ DEF_OP(StoreNZCV) {
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value.ID()));
}
DEF_OP(LoadDF) {
auto Dst = GetReg(Node);
auto Flag = X86State::RFLAG_DF_RAW_LOC;
// DF needs sign extension to turn 0x1/0xFF into 1/-1
ldrsb(Dst.X(), STATE, offsetof(FEXCore::Core::CPUState, flags[Flag]));
}
DEF_OP(LoadFlag) {
auto Op = IROp->C<IR::IROp_LoadFlag>();
auto Dst = GetReg(Node);
@@ -1174,10 +1330,8 @@ DEF_OP(LoadMemTSO) {
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
break;
}
if (VectorTSOEnabled()) {
// Half-barrier.
dmb(FEXCore::ARMEmitter::BarrierScope::ISHLD);
}
// Half-barrier.
dmb(FEXCore::ARMEmitter::BarrierScope::ISHLD);
}
}
@@ -1325,7 +1479,7 @@ DEF_OP(VLoadVectorElement) {
}
// Emit a half-barrier if TSO is enabled.
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
if (CTX->IsAtomicTSOEnabled()) {
dmb(ARMEmitter::BarrierScope::ISHLD);
}
}
@@ -1345,7 +1499,7 @@ DEF_OP(VStoreVectorElement) {
ElementSize == 16, "Invalid element size");
// Emit a half-barrier if TSO is enabled.
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
if (CTX->IsAtomicTSOEnabled()) {
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
}
@@ -1445,7 +1599,7 @@ DEF_OP(VBroadcastFromMem) {
}
// Emit a half-barrier if TSO is enabled.
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
if (CTX->IsAtomicTSOEnabled()) {
dmb(ARMEmitter::BarrierScope::ISHLD);
}
}
@@ -1663,10 +1817,8 @@ DEF_OP(StoreMemTSO) {
}
}
else {
if (VectorTSOEnabled()) {
// Half-Barrier.
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
}
// Half-Barrier.
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
const auto Src = GetVReg(Op->Value.ID());
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
@@ -1707,7 +1859,6 @@ DEF_OP(MemSet) {
// that the value is zero, we can optimize any operation larger than 8-bit down to 8-bit to use the MOPS implementation.
const auto Op = IROp->C<IR::IROp_MemSet>();
const bool IsAtomic = Op->IsAtomic && MemcpySetTSOEnabled();
const int32_t Size = Op->Size;
const auto MemReg = GetReg(Op->Addr.ID());
const auto Value = GetReg(Op->Value.ID());
@@ -1721,15 +1872,15 @@ DEF_OP(MemSet) {
DirectionReg = GetReg(Op->Direction.ID());
}
// If Direction > 0 then:
// If Direction == 0 then:
// MemReg is incremented (by size)
// else:
// MemReg is decremented (by size)
//
// Counter is decremented regardless.
ARMEmitter::SingleUseForwardLabel BackwardImpl{};
ARMEmitter::SingleUseForwardLabel Done{};
ARMEmitter::ForwardLabel BackwardImpl{};
ARMEmitter::ForwardLabel Done{};
mov(TMP1, Length.X());
if (Op->Prefix.IsInvalid()) {
@@ -1742,7 +1893,7 @@ DEF_OP(MemSet) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
tbnz(DirectionReg, 1, &BackwardImpl);
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
}
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
@@ -1797,73 +1948,18 @@ DEF_OP(MemSet) {
}
};
const auto SubRegSize =
Size == 1 ? ARMEmitter::SubRegSize::i8Bit :
Size == 2 ? ARMEmitter::SubRegSize::i16Bit :
Size == 4 ? ARMEmitter::SubRegSize::i32Bit :
Size == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
auto EmitMemset = [&](int32_t Direction) {
const int32_t OpSize = Size;
const int32_t SizeDirection = Size * Direction;
ARMEmitter::BiDirectionalLabel AgainInternal{};
ARMEmitter::BackwardLabel AgainInternal{};
ARMEmitter::ForwardLabel DoneInternal{};
// Early exit if zero count.
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AgainInternal256Exit{};
ARMEmitter::BackwardLabel AgainInternal256{};
ARMEmitter::ForwardLabel AgainInternal128Exit{};
ARMEmitter::BackwardLabel AgainInternal128{};
if (Direction == -1) {
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
}
// Keep the counter one copy ahead, so that underflow can be used to detect when to fallback
// to the copy unit size copy loop for the last chunk.
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal128Exit);
// Fill VTMP2 with the set pattern
dup(SubRegSize, VTMP2.Q(), Value);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal256Exit);
Bind(&AgainInternal256);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
tbz(TMP1, 63, &AgainInternal256);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbz(TMP1, 63, &AgainInternal128);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
}
}
Bind(&AgainInternal);
if (IsAtomic) {
if (Op->IsAtomic) {
MemStoreTSO(Value, OpSize, SizeDirection);
}
else {
@@ -1915,8 +2011,8 @@ DEF_OP(MemSet) {
};
if (DirectionIsInline) {
LOGMAN_THROW_AA_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
EmitMemset(DirectionConstant);
// If the direction constant is set then the direction is negative.
EmitMemset(DirectionConstant ? -1 : 1);
}
else {
// Emit forward direction memset then backward direction memset.
@@ -1941,7 +2037,6 @@ DEF_OP(MemCpy) {
// Assuming non-atomicity and non-faulting behaviour, this can accelerate this implementation.
const auto Op = IROp->C<IR::IROp_MemCpy>();
const bool IsAtomic = Op->IsAtomic && MemcpySetTSOEnabled();
const int32_t Size = Op->Size;
const auto MemRegDest = GetReg(Op->AddrDest.ID());
const auto MemRegSrc = GetReg(Op->AddrSrc.ID());
@@ -1955,7 +2050,7 @@ DEF_OP(MemCpy) {
}
auto Dst = GetRegPair(Node);
// If Direction > 0 then:
// If Direction == 0 then:
// MemRegDest is incremented (by size)
// MemRegSrc is incremented (by size)
// else:
@@ -1964,8 +2059,8 @@ DEF_OP(MemCpy) {
//
// Counter is decremented regardless.
ARMEmitter::SingleUseForwardLabel BackwardImpl{};
ARMEmitter::SingleUseForwardLabel Done{};
ARMEmitter::ForwardLabel BackwardImpl{};
ARMEmitter::ForwardLabel Done{};
mov(TMP1, Length.X());
if (Op->PrefixDest.IsInvalid()) {
@@ -1991,7 +2086,7 @@ DEF_OP(MemCpy) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
tbnz(DirectionReg, 1, &BackwardImpl);
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
}
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
@@ -2012,10 +2107,6 @@ DEF_OP(MemCpy) {
ldr<ARMEmitter::IndexType::POST>(TMP4, TMP3, Size);
str<ARMEmitter::IndexType::POST>(TMP4, TMP2, Size);
break;
case 32:
ldp<ARMEmitter::IndexType::POST>(VTMP1.Q(), VTMP2.Q(), TMP3, Size);
stp<ARMEmitter::IndexType::POST>(VTMP1.Q(), VTMP2.Q(), TMP2, Size);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
break;
@@ -2122,69 +2213,14 @@ DEF_OP(MemCpy) {
const int32_t OpSize = Size;
const int32_t SizeDirection = Size * Direction;
ARMEmitter::BiDirectionalLabel AgainInternal{};
ARMEmitter::BackwardLabel AgainInternal{};
ARMEmitter::ForwardLabel DoneInternal{};
// Early exit if zero count.
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AbsPos{};
ARMEmitter::ForwardLabel AgainInternal256Exit{};
ARMEmitter::ForwardLabel AgainInternal128Exit{};
ARMEmitter::BackwardLabel AgainInternal128{};
ARMEmitter::BackwardLabel AgainInternal256{};
sub(ARMEmitter::Size::i64Bit, TMP4, TMP2, TMP3);
tbz(TMP4, 63, &AbsPos);
neg(ARMEmitter::Size::i64Bit, TMP4, TMP4);
Bind(&AbsPos);
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
tbnz(TMP4, 63, &AgainInternal);
if (Direction == -1) {
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
sub(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32 - Size);
}
// Keep the counter one copy ahead, so that underflow can be used to detect when to fallback
// to the copy unit size copy loop for the last chunk.
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal128Exit);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal256Exit);
Bind(&AgainInternal256);
MemCpy(32, 32 * Direction);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
tbz(TMP1, 63, &AgainInternal256);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
tbz(TMP1, 63, &AgainInternal128);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
add(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32 - Size);
}
}
Bind(&AgainInternal);
if (IsAtomic) {
if (Op->IsAtomic) {
MemCpyTSO(OpSize, SizeDirection);
}
else {
@@ -2249,8 +2285,8 @@ DEF_OP(MemCpy) {
};
if (DirectionIsInline) {
LOGMAN_THROW_AA_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
EmitMemcpy(DirectionConstant);
// If the direction constant is set then the direction is negative.
EmitMemcpy(DirectionConstant ? -1 : 1);
}
else {
// Emit forward direction memset then backward direction memset.
@@ -2547,47 +2583,6 @@ DEF_OP(CacheLineZero) {
}
}
DEF_OP(Prefetch) {
auto Op = IROp->C<IR::IROp_Prefetch>();
const auto MemReg = GetReg(Op->Addr.ID());
// Access size is only ever handled as 8-byte. Even though it is accesssed as a cacheline.
const auto MemSrc = GenerateMemOperand(8, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
size_t LUT =
(Op->Stream ? 1 : 0) |
((Op->CacheLevel - 1) << 1) |
(Op->ForStore ? 1U << 3 : 0);
constexpr static std::array<ARMEmitter::Prefetch, 14> PrefetchType = {
ARMEmitter::Prefetch::PLDL1KEEP,
ARMEmitter::Prefetch::PLDL1STRM,
ARMEmitter::Prefetch::PLDL2KEEP,
ARMEmitter::Prefetch::PLDL2STRM,
ARMEmitter::Prefetch::PLDL3KEEP,
ARMEmitter::Prefetch::PLDL3STRM,
// Gap of two.
// 0b0'11'0
ARMEmitter::Prefetch::PLDL1STRM,
// 0b0'11'1
ARMEmitter::Prefetch::PLDL1STRM,
ARMEmitter::Prefetch::PSTL1KEEP,
ARMEmitter::Prefetch::PSTL1STRM,
ARMEmitter::Prefetch::PSTL2KEEP,
ARMEmitter::Prefetch::PSTL2STRM,
ARMEmitter::Prefetch::PSTL3KEEP,
ARMEmitter::Prefetch::PSTL3STRM,
};
prfm(PrefetchType[LUT], MemSrc);
}
#undef DEF_OP
}
@@ -1372,32 +1372,6 @@ DEF_OP(VUMinV) {
}
}
DEF_OP(VUMaxV) {
const auto Op = IROp->C<IR::IROp_VUMaxV>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto ElementSize = Op->Header.ElementSize;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize =
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
if (HostSupportsSVE256 && Is256Bit) {
const auto Pred = PRED_TMP_32B;
umaxv(SubRegSize, Dst, Pred, Vector.Z());
} else {
// Vector
umaxv(SubRegSize, Dst.Q(), Vector.Q());
}
}
DEF_OP(VURAvg) {
const auto Op = IROp->C<IR::IROp_VURAvg>();
const auto OpSize = IROp->Size;
@@ -5035,50 +5009,42 @@ DEF_OP(VTBX1) {
if (Dst != VectorSrcDst) {
switch (OpSize) {
case 8: {
mov(VTMP1.D(), VectorSrcDst.D());
tbx(VTMP1.D(), VectorTable.Q(), VectorIndices.D());
mov(Dst.D(), VTMP1.D());
mov(Dst.D(), VectorSrcDst.D());
break;
}
case 16: {
mov(VTMP1.Q(), VectorSrcDst.Q());
tbx(VTMP1.Q(), VectorTable.Q(), VectorIndices.Q());
mov(Dst.Q(), VTMP1.Q());
mov(Dst.Q(), VectorSrcDst.Q());
break;
}
case 32: {
LOGMAN_THROW_AA_FMT(HostSupportsSVE256,
"Host does not support SVE. Cannot perform 256-bit table lookup");
mov(VTMP1.Z(), VectorSrcDst.Z());
tbx(ARMEmitter::SubRegSize::i8Bit, VTMP1.Z(), VectorTable.Z(), VectorIndices.Z());
mov(Dst.Z(), VTMP1.Z());
mov(Dst.Z(), VectorSrcDst.Z());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown OpSize: {}", OpSize);
break;
}
} else {
switch (OpSize) {
case 8: {
tbx(VectorSrcDst.D(), VectorTable.Q(), VectorIndices.D());
break;
}
case 16: {
tbx(VectorSrcDst.Q(), VectorTable.Q(), VectorIndices.Q());
break;
}
case 32: {
LOGMAN_THROW_AA_FMT(HostSupportsSVE256,
"Host does not support SVE. Cannot perform 256-bit table lookup");
}
tbx(ARMEmitter::SubRegSize::i8Bit, VectorSrcDst.Z(), VectorTable.Z(), VectorIndices.Z());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown OpSize: {}", OpSize);
break;
switch (OpSize) {
case 8: {
tbx(Dst.D(), VectorTable.Q(), VectorIndices.D());
break;
}
case 16: {
tbx(Dst.Q(), VectorTable.Q(), VectorIndices.Q());
break;
}
case 32: {
LOGMAN_THROW_AA_FMT(HostSupportsSVE256,
"Host does not support SVE. Cannot perform 256-bit table lookup");
tbx(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), VectorTable.Z(), VectorIndices.Z());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown OpSize: {}", OpSize);
break;
}
}
+5 -1
View File
@@ -1,7 +1,7 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Interface/Core/CPUBackend.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/fextl/memory.h>
namespace FEXCore::Context {
@@ -15,6 +15,10 @@ struct InternalThreadState;
namespace FEXCore::CPU {
class CPUBackend;
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx,
FEXCore::Core::InternalThreadState *Thread);
CPUBackendFeatures GetX86JITBackendFeatures();
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx,
FEXCore::Core::InternalThreadState *Thread);
CPUBackendFeatures GetArm64JITBackendFeatures();
+9 -8
View File
@@ -100,15 +100,15 @@ public:
L1Entry.HostCode = (uintptr_t)HostCode;
}
void Erase(FEXCore::Core::CpuStateFrame *Frame, uint64_t Address) {
void Erase(uint64_t Address) {
std::lock_guard<std::recursive_mutex> lk(WriteLock);
// Sever any links to this block
auto lower = BlockLinks->lower_bound({Address, nullptr});
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData *>(UINTPTR_MAX)});
auto lower = BlockLinks->lower_bound({Address, 0});
auto upper = BlockLinks->upper_bound({Address, UINTPTR_MAX});
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
it->second(Frame, it->first.HostLink);
it->second();
}
// Remove from BlockList
@@ -141,7 +141,8 @@ public:
BlockPointers[PageOffset].HostCode = 0;
}
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData * HostLink, const FEXCore::Context::BlockDelinkerFunc &delinker) {
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
std::lock_guard<std::recursive_mutex> lk(WriteLock);
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
@@ -157,7 +158,7 @@ public:
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
// This needs to be taken before reads or writes to L2, L3, CodePages,
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
// may only happen during cross thread invalidation (::Erase).
// All other operations must be done from the owning thread.
@@ -223,7 +224,7 @@ private:
struct BlockLinkTag {
uint64_t GuestDestination;
FEXCore::Context::ExitFunctionLinkData *HostLink;
uintptr_t HostLink;
bool operator <(const BlockLinkTag& other) const {
if (GuestDestination < other.GuestDestination)
@@ -242,7 +243,7 @@ private:
//
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
using BlockLinksMapType = std::pmr::map<BlockLinkTag, std::function<void()>>;
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
BlockLinksMapType *BlockLinks;
@@ -32,6 +32,9 @@ namespace FEXCore::CodeSerialize {
// ABI local flag unsafe optimization
unsigned ABILocalFlags : 1;
// Static register allocation enabled
unsigned SRA : 1;
// Paranoid TSO mode enabled
unsigned ParanoidTSO : 1;
@@ -46,7 +49,7 @@ namespace FEXCore::CodeSerialize {
// Padding to remove uninitialized data warning from asan
// Shows remaining amount of bits available for config
unsigned _Pad : 19;
unsigned _Pad : 18;
bool operator==(CodeObjectSerializationConfig const &other) const {
return Cookie == other.Cookie &&
@@ -56,6 +59,7 @@ namespace FEXCore::CodeSerialize {
HardwareTSOEnabled == other.HardwareTSOEnabled &&
TSOEnabled == other.TSOEnabled &&
ABILocalFlags == other.ABILocalFlags &&
SRA == other.SRA &&
ParanoidTSO == other.ParanoidTSO &&
Is64BitMode == other.Is64BitMode &&
SMCChecks == other.SMCChecks &&
@@ -71,6 +75,7 @@ namespace FEXCore::CodeSerialize {
Hash <<= 1; Hash |= other.HardwareTSOEnabled;
Hash <<= 1; Hash |= other.TSOEnabled;
Hash <<= 1; Hash |= other.ABILocalFlags;
Hash <<= 1; Hash |= other.SRA;
Hash <<= 1; Hash |= other.ParanoidTSO;
Hash <<= 1; Hash |= other.Is64BitMode;
Hash <<= 2; Hash |= other.SMCChecks;
@@ -18,6 +18,7 @@ namespace FEXCore::CodeSerialize {
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
DefaultSerializationConfig.SRA = ctx->Config.StaticRegisterAllocation;
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
File diff suppressed because it is too large. Load diff
+207 -153
View File
@@ -4,12 +4,13 @@
#include "Interface/Core/Frontend.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include "Interface/Context/Context.h"
#include "Interface/IR/IREmitter.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
@@ -77,7 +78,10 @@ friend class FEXCore::IR::PassManager;
public:
enum class FlagsGenerationType : uint8_t {
TYPE_NONE,
TYPE_ADC,
TYPE_SBB,
TYPE_SUB,
TYPE_ADD,
TYPE_MUL,
TYPE_UMUL,
TYPE_LOGICAL,
@@ -88,13 +92,18 @@ public:
TYPE_LSHRDI,
TYPE_ASHR,
TYPE_ASHRI,
TYPE_ROR,
TYPE_RORI,
TYPE_ROL,
TYPE_ROLI,
TYPE_BEXTR,
TYPE_BLSI,
TYPE_BLSMSK,
TYPE_BLSR,
TYPE_POPCOUNT,
TYPE_BZHI,
TYPE_ZCNT,
TYPE_TZCNT,
TYPE_LZCNT,
TYPE_RDRAND,
};
@@ -223,32 +232,6 @@ public:
return CanHaveSideEffects;
}
template <typename F>
void ForeachDirection(F&& Routine) {
// Otherwise, prepare to branch.
auto Zero = _Constant(0);
// If the shift is zero, do not touch the flags.
auto ForwardBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
auto BackwardBlock = CreateNewCodeBlockAfter(ForwardBlock);
auto ExitBlock = CreateNewCodeBlockAfter(BackwardBlock);
auto DF = GetRFLAG(X86State::RFLAG_DF_RAW_LOC);
CondJump(DF, Zero, ForwardBlock, BackwardBlock, {COND_EQ});
for (auto D = 0; D < 2; ++D) {
SetCurrentCodeBlock(D ? BackwardBlock : ForwardBlock);
StartNewBlock();
{
Routine(D ? -1 : 1);
Jump(ExitBlock);
}
}
SetCurrentCodeBlock(ExitBlock);
StartNewBlock();
}
OpDispatchBuilder(FEXCore::Context::ContextImpl *ctx);
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
@@ -278,7 +261,6 @@ public:
template<FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp>
void ALUOp(OpcodeArgs);
void INTOp(OpcodeArgs);
template<bool IsSyscallInst>
void SyscallOp(OpcodeArgs);
void ThunkOp(OpcodeArgs);
void LEAOp(OpcodeArgs);
@@ -289,9 +271,8 @@ public:
void SecondaryALUOp(OpcodeArgs);
template<uint32_t SrcIndex>
void ADCOp(OpcodeArgs);
template<uint32_t SrcIndex>
template<uint32_t SrcIndex, bool SetFlags>
void SBBOp(OpcodeArgs);
void SALCOp(OpcodeArgs);
void PUSHOp(OpcodeArgs);
void PUSHREGOp(OpcodeArgs);
void PUSHAOp(OpcodeArgs);
@@ -329,22 +310,25 @@ public:
void CMOVOp(OpcodeArgs);
void CPUIDOp(OpcodeArgs);
void XGetBVOp(OpcodeArgs);
uint32_t LoadConstantShift(X86Tables::DecodedOp Op, bool Is1Bit);
void SHLOp(OpcodeArgs);
template<bool SHL1Bit>
void SHLOp(OpcodeArgs);
void SHLImmediateOp(OpcodeArgs);
void SHROp(OpcodeArgs);
template<bool SHR1Bit>
void SHROp(OpcodeArgs);
void SHRImmediateOp(OpcodeArgs);
void SHLDOp(OpcodeArgs);
void SHLDImmediateOp(OpcodeArgs);
void SHRDOp(OpcodeArgs);
void SHRDImmediateOp(OpcodeArgs);
void ASHROp(OpcodeArgs);
template<bool SHR1Bit>
void ASHROp(OpcodeArgs);
void ASHRImmediateOp(OpcodeArgs);
template<bool Left, bool IsImmediate, bool Is1Bit>
void RotateOp(OpcodeArgs);
template<bool Is1Bit>
void ROROp(OpcodeArgs);
void RORImmediateOp(OpcodeArgs);
template<bool Is1Bit>
void ROLOp(OpcodeArgs);
void ROLImmediateOp(OpcodeArgs);
void RCROp1Bit(OpcodeArgs);
void RCROp8x1Bit(OpcodeArgs);
void RCROp(OpcodeArgs);
@@ -368,11 +352,6 @@ public:
void PUSHFOp(OpcodeArgs);
void POPFOp(OpcodeArgs);
struct CycleCounterPair {
OrderedNode *CounterLow;
OrderedNode *CounterHigh;
};
CycleCounterPair CycleCounter();
void RDTSCOp(OpcodeArgs);
void INCOp(OpcodeArgs);
void DECOp(OpcodeArgs);
@@ -827,7 +806,6 @@ public:
void FXSaveOp(OpcodeArgs);
void FXRStoreOp(OpcodeArgs);
OrderedNode *XSaveBase(X86Tables::DecodedOp Op);
void XSaveOp(OpcodeArgs);
void PAlignrOp(OpcodeArgs);
@@ -886,15 +864,9 @@ public:
void StoreFenceOrCLFlush(OpcodeArgs);
void CLZeroOp(OpcodeArgs);
void RDTSCPOp(OpcodeArgs);
void RDPIDOp(OpcodeArgs);
template<bool ForStore, bool Stream, uint8_t Level>
void Prefetch(OpcodeArgs);
void PSADBW(OpcodeArgs);
OrderedNode *BitwiseAtLeastTwo(OrderedNode *A, OrderedNode *B, OrderedNode *C);
void SHA1NEXTEOp(OpcodeArgs);
void SHA1MSG1Op(OpcodeArgs);
void SHA1MSG2Op(OpcodeArgs);
@@ -1000,7 +972,16 @@ private:
}
static bool IsNZCV(unsigned BitOffset) {
return ContainsNZCV(1U << BitOffset);
switch (BitOffset) {
case FEXCore::X86State::RFLAG_CF_RAW_LOC:
case FEXCore::X86State::RFLAG_ZF_RAW_LOC:
case FEXCore::X86State::RFLAG_SF_RAW_LOC:
case FEXCore::X86State::RFLAG_OF_RAW_LOC:
return true;
default:
return false;
}
}
OrderedNode* CachedNZCV{};
@@ -1014,7 +995,7 @@ private:
// Used during new op bringup
bool ShouldDump{false};
void ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
void ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp);
// Opcode helpers for generalizing behavior across VEX and non-VEX variants.
@@ -1280,34 +1261,14 @@ private:
return NZCVMask;
}
// Set flag tracking to prepare for an operation that directly writes NZCV. If
// some bits are known to be zeroed, the PossiblySetNZCVBits mask can be
// passed. Otherwise, it defaults to assuming all bits may be set after
// (this is conservative).
void HandleNZCVWrite(uint32_t _PossiblySetNZCVBits = ~0) {
InvalidateDeferredFlags();
CachedNZCV = nullptr;
PossiblySetNZCVBits = _PossiblySetNZCVBits;
NZCVDirty = false;
}
// Set flag tracking to prepare for a read-modify-write operation on NZCV.
void HandleNZCV_RMW(uint32_t _PossiblySetNZCVBits = ~0) {
if (NZCVDirty && CachedNZCV)
_StoreNZCV(CachedNZCV);
HandleNZCVWrite(_PossiblySetNZCVBits);
}
// Special case of the above where we are known to zero C/V
void HandleNZ00Write() {
HandleNZCVWrite((1u << 31) | (1u << 30));
}
OrderedNode *GetNZCV() {
if (!CachedNZCV)
if (!CachedNZCV) {
CachedNZCV = _LoadNZCV();
// We don't know what's set
PossiblySetNZCVBits = ~0;
}
return CachedNZCV;
}
@@ -1334,8 +1295,10 @@ private:
}
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
HandleNZ00Write();
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
CachedNZCV = _LoadNZCV();
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
NZCVDirty = false;
}
void InsertNZCV(unsigned BitOffset, OrderedNode *Value, signed FlagOffset, bool MustMask) {
@@ -1409,11 +1372,6 @@ private:
if (ValueOffset || MustMask)
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
// For DF, we need to transform 0/1 into 1/-1
if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
Value = _SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1);
}
_StoreFlag(Value, BitOffset);
}
}
@@ -1426,7 +1384,7 @@ private:
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(Constant << 4));
}
void ZeroPF_AF();
void ZeroMultipleFlags(uint32_t BitMask);
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
switch (BitOffset) {
@@ -1466,32 +1424,11 @@ private:
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
// Recover the sign bit, it is the logical DF value
return _Lshr(OpSize::i64Bit, _LoadDF(), _Constant(63));
} else {
return _LoadFlag(BitOffset);
}
}
// Returns (DF ? -Size : Size)
OrderedNode *LoadDir(const unsigned Size) {
auto Dir = _LoadDF();
auto Shift = FEXCore::ilog2(Size);
if (Shift)
return _Lshl(IR::SizeToOpSize(CTX->GetGPRSize()), Dir, _Constant(Shift));
else
return Dir;
}
// Returns DF ? (X - Size) : (X + Size)
OrderedNode *OffsetByDir(OrderedNode *X, const unsigned Size) {
auto Shift = FEXCore::ilog2(Size);
return _AddShift(OpSize::i64Bit, X, _LoadDF(), ShiftType::LSL, Shift);
}
// Set SSE comparison flags based on the result set by Arm FCMP. This converts
// NZCV from the Arm representation to an eXternal representation that's
// totally not a euphemism for x86 or anything, nuh-uh.
@@ -1631,7 +1568,7 @@ private:
}
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP) const;
OrderedNode *SelectBit(OrderedNode *Cmp, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
OrderedNode *SelectBit(OrderedNode *Cmp, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
OrderedNode *SelectCC(uint8_t OP, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
/**
@@ -1664,22 +1601,29 @@ private:
OrderedNode *Res{};
union {
// UMUL, BEXTR, BLSI, POPCOUNT, ZCNT, RDRAND
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, RDRAND
struct {
} NoSource;
// MUL, BLSR, BLSMSKB, BZHI
// MUL, BLSR, BZHI
struct {
OrderedNode *Src1;
} OneSource;
// Logical, LSHL, LSHR, ASHR
// Logical, LSHL, LSHR, ASHR, ROR, ROL
struct {
OrderedNode *Src1;
OrderedNode *Src2;
} TwoSource;
// LSHLI, LSHRI, ASHRI
// ADC, SBB
struct {
OrderedNode *Src1;
OrderedNode *Src2;
OrderedNode *Src3;
} ThreeSource;
// LSHLI, LSHRI, ASHRI, RORI, ROLI
struct {
OrderedNode *Src1;
uint64_t Imm;
@@ -1728,12 +1672,15 @@ private:
}
template <typename F>
void Calculate_ShiftVariable(OrderedNode *Shift, F&& Calculate) {
void CalculateFlags_ShiftVariable(OrderedNode *Shift, F&& CalculateFlags) {
// We are the ones calculating the deferred flags. Don't recurse!
InvalidateDeferredFlags();
// RCR can call this with constants, so handle that without branching.
uint64_t Const;
if (IsValueConstant(WrapNode(Shift), &Const)) {
if (Const)
Calculate();
CalculateFlags();
return;
}
@@ -1750,7 +1697,7 @@ private:
SetCurrentCodeBlock(SetBlock);
StartNewBlock();
{
Calculate();
CalculateFlags();
Jump(EndBlock);
}
@@ -1759,29 +1706,20 @@ private:
PossiblySetNZCVBits |= OldSetNZCVBits;
}
template <typename F>
void CalculateFlags_ShiftVariable(OrderedNode *Shift, F&& CalculateFlags) {
// We are the ones calculating the deferred flags. Don't recurse!
InvalidateDeferredFlags();
Calculate_ShiftVariable(Shift, CalculateFlags);
}
/**
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
* @{ */
OrderedNode *LoadPFRaw(bool Invert);
OrderedNode *LoadPFRaw();
OrderedNode *LoadAF();
void FixupAF();
void SetAFAndFixup(OrderedNode *AF);
OrderedNode *CalculateAFForDecimal(OrderedNode *A);
void CalculatePF(OrderedNode *Res);
void CalculateAF(OrderedNode *Src1, OrderedNode *Src2);
void CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool Sub);
OrderedNode *CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2);
OrderedNode *CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2);
OrderedNode *CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
OrderedNode *CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
void CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
void CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
void CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
void CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
void CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High);
void CalculateFlags_UMUL(OrderedNode *High);
void CalculateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
@@ -1793,13 +1731,18 @@ private:
void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
void CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
void CalculateFlags_BEXTR(OrderedNode *Src);
void CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
void CalculateFlags_BLSMSK(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
void CalculateFlags_BLSMSK(OrderedNode *Src);
void CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
void CalculateFlags_POPCOUNT(OrderedNode *Src);
void CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
void CalculateFlags_ZCNT(uint8_t SrcSize, OrderedNode *Result);
void CalculateFlags_TZCNT(OrderedNode *Src);
void CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
void CalculateFlags_RDRAND(OrderedNode *Src);
/** @} */
@@ -1808,7 +1751,37 @@ private:
*
* Depending on the operation it may force a RFLAGs calculation before storing the new deferred state.
* @{ */
void GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true) {
void GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_ADC,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.ThreeSource = {
.Src1 = Src1,
.Src2 = Src2,
.Src3 = CF,
},
},
};
}
void GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_SBB,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.ThreeSource = {
.Src1 = Src1,
.Src2 = Src2,
.Src3 = CF,
},
},
};
}
void GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true) {
if (!UpdateCF) {
// If we aren't updating CF then we need to calculate flags. Invalidation mask would make this not required.
CalculateDeferredFlags();
@@ -1816,6 +1789,26 @@ private:
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_SUB,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.TwoSrcImmediate = {
.Src1 = Src1,
.Src2 = Src2,
.UpdateCF = UpdateCF,
},
},
};
}
void GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true) {
if (!UpdateCF) {
// If we aren't updating CF then we need to calculate flags. Invalidation mask would make this not required.
CalculateDeferredFlags();
}
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_ADD,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.TwoSrcImmediate = {
.Src1 = Src1,
@@ -1980,6 +1973,78 @@ private:
};
}
void GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
// Doesn't set all the flags, needs to calculate.
CalculateDeferredFlags();
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_ROR,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.TwoSource = {
.Src1 = Src1,
.Src2 = Src2,
},
},
};
}
void GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
// Doesn't set all the flags, needs to calculate.
CalculateDeferredFlags();
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_ROL,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.TwoSource = {
.Src1 = Src1,
.Src2 = Src2,
},
},
};
}
void GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
if (Shift == 0) return;
// Doesn't set all the flags, needs to calculate.
CalculateDeferredFlags();
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_RORI,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.OneSrcImmediate = {
.Src1 = Src1,
.Imm = Shift,
},
},
};
}
void GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
if (Shift == 0) return;
// Doesn't set all the flags, needs to calculate.
CalculateDeferredFlags();
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_ROLI,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.OneSrcImmediate = {
.Src1 = Src1,
.Imm = Shift,
},
}
};
}
void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_BEXTR,
@@ -1996,16 +2061,11 @@ private:
};
}
void GenerateFlags_BLSMSK(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src) {
void GenerateFlags_BLSMSK(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_BLSMSK,
.SrcSize = GetSrcSize(Op),
.Res = Res,
.Sources = {
.OneSource = {
.Src1 = Src,
},
},
.Res = Src,
};
}
@@ -2043,9 +2103,17 @@ private:
};
}
void GenerateFlags_ZCNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
void GenerateFlags_TZCNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_ZCNT,
.Type = FlagsGenerationType::TYPE_TZCNT,
.SrcSize = GetSrcSize(Op),
.Res = Src,
};
}
void GenerateFlags_LZCNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_LZCNT,
.SrcSize = GetSrcSize(Op),
.Res = Src,
};
@@ -2059,16 +2127,6 @@ private:
};
}
OrderedNode *AndConst(FEXCore::IR::OpSize Size, OrderedNode *Node, uint64_t Const) {
uint64_t NodeConst;
if (IsValueConstant(WrapNode(Node), &NodeConst)) {
return _Constant(NodeConst & Const);
} else {
return _And(Size, Node, _Constant(Const));
}
}
/** @} */
/** @} */
@@ -2109,10 +2167,6 @@ private:
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
OrderedNode* Prefetch(bool ForStore, bool Stream, uint8_t CacheLevel, OrderedNode *ssa0) {
return _Prefetch(ForStore, Stream, CacheLevel, ssa0, Invalid(), MEM_OFFSET_SXTX, 1);
}
void InstallHostSpecificOpcodeHandlers();
///< Segment telemetry tracking
@@ -8,6 +8,7 @@ $end_info$
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Core/OpcodeDispatcher.h"
@@ -103,7 +104,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
};
const auto f2 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self.BitwiseAtLeastTwo(B, C, D);
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._And(OpSize::i32Bit, B, D)), Self._And(OpSize::i32Bit, C, D));
};
const auto f3 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
@@ -128,6 +129,9 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto W0E = _VExtractToGPR(16, 4, Src, 3);
auto W1 = _VExtractToGPR(16, 4, Src, 2);
auto W2 = _VExtractToGPR(16, 4, Src, 1);
auto W3 = _VExtractToGPR(16, 4, Src, 0);
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
@@ -146,12 +150,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
return {A1, B1, C1, D1, E1};
};
const auto Round1To3 = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C,
OrderedNode *D, OrderedNode *E, OrderedNode *Src, unsigned W_idx) -> RoundResult {
// Kill W and E at the beginning
auto W = _VExtractToGPR(16, 4, Src, W_idx);
auto Q = _Add(OpSize::i32Bit, W, E);
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), Q), K);
OrderedNode *D, OrderedNode *E, OrderedNode *W) -> RoundResult {
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W), E), K);
auto BNext = A;
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
auto DNext = C;
@@ -161,9 +161,9 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
};
auto [A1, B1, C1, D1, E1] = Round0();
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, W1);
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, W2);
auto Final = Round1To3(A3, B3, C3, D3, E3, W3);
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
auto Dest2 = _VInsGPR(16, 4, 2, Dest3, std::get<1>(Final));
@@ -230,25 +230,18 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
StoreResult(FPRClass, Op, D0, -1);
}
OrderedNode *OpDispatchBuilder::BitwiseAtLeastTwo(OrderedNode *A, OrderedNode *B, OrderedNode *C) {
// Returns whether at least 2/3 of A/B/C is true.
// Expressed as (A & (B | C)) | (B & C)
//
// Equivalent to expression in SHA calculations: (A & B) ^ (A & C) ^ (B & C)
auto And = _And(OpSize::i32Bit, B, C);
auto Or = _Or(OpSize::i32Bit, B, C);
return _Or(OpSize::i32Bit, _And(OpSize::i32Bit, A, Or), And);
}
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
const auto Ch = [this](OrderedNode *E, OrderedNode *F, OrderedNode *G) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
};
const auto Major = [this](OrderedNode *A, OrderedNode *B, OrderedNode *C) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, A, B), _And(OpSize::i32Bit, A, C)), _And(OpSize::i32Bit, B, C));
};
const auto Sigma0 = [this](OrderedNode *A) -> OrderedNode* {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), A, ShiftType::ROR, 13), A, ShiftType::ROR, 22);
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), _Ror(OpSize::i32Bit, A, _Constant(32, 13))), _Ror(OpSize::i32Bit, A, _Constant(32, 22)));
};
const auto Sigma1 = [this](OrderedNode *E) -> OrderedNode* {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), E, ShiftType::ROR, 11), E, ShiftType::ROR, 25);
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), _Ror(OpSize::i32Bit, E, _Constant(32, 11))), _Ror(OpSize::i32Bit, E, _Constant(32, 25)));
};
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
@@ -256,44 +249,42 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
// Hardcoded to XMM0
auto XMM0 = LoadXMMRegister(0);
auto E0 = _VExtractToGPR(16, 4, Src, 1);
auto F0 = _VExtractToGPR(16, 4, Src, 0);
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
OrderedNode *Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
Q0 = _Add(OpSize::i32Bit, Q0, H0);
auto A0 = _VExtractToGPR(16, 4, Src, 3);
auto B0 = _VExtractToGPR(16, 4, Src, 2);
auto C0 = _VExtractToGPR(16, 4, Dest, 3);
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
auto D0 = _VExtractToGPR(16, 4, Dest, 2);
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
OrderedNode * Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
auto E0 = _VExtractToGPR(16, 4, Src, 1);
auto F0 = _VExtractToGPR(16, 4, Src, 0);
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
auto WK1 = _VExtractToGPR(16, 4, XMM0, 1);
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
G0 = _VExtractToGPR(16, 4, Dest, 1);
Q1 = _Add(OpSize::i32Bit, Q1, G0);
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*,
OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
const auto Round = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C, OrderedNode *D,
OrderedNode *E, OrderedNode *F, OrderedNode *G, OrderedNode *H,
OrderedNode* WK) -> RoundResult {
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Ch(E, F, G), Sigma1(E)), WK), H), Major(A, B, C)), Sigma0(A));
auto BNext = A;
auto CNext = B;
auto DNext = C;
auto ENext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Ch(E, F, G), Sigma1(E)), WK), H), D);
auto FNext = E;
auto GNext = F;
auto HNext = G;
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
return {ANext, BNext, CNext, DNext, ENext, FNext, GNext, HNext};
};
// Rematerialize C0. As with G0.
C0 = _VExtractToGPR(16, 4, Dest, 3);
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
auto Res3 = _VInsGPR(16, 4, 3, Dest, A2);
auto Res2 = _VInsGPR(16, 4, 2, Res3, A1);
auto Res1 = _VInsGPR(16, 4, 1, Res2, E2);
auto Res0 = _VInsGPR(16, 4, 0, Res1, E1);
auto [A1, B1, C1, D1, E1, F1, G1, H1] = Round(A0, B0, C0, D0, E0, F0, G0, H0, WK0);
auto Final = Round(A1, B1, C1, D1, E1, F1, G1, H1, WK1);
auto Res3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
auto Res2 = _VInsGPR(16, 4, 2, Res3, std::get<1>(Final));
auto Res1 = _VInsGPR(16, 4, 1, Res2, std::get<4>(Final));
auto Res0 = _VInsGPR(16, 4, 0, Res1, std::get<5>(Final));
StoreResult(FPRClass, Op, Res0, -1);
}
@@ -13,6 +13,7 @@ $end_info$
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/IR/IR.h>
#include <array>
#include <cstdint>
@@ -26,7 +27,7 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
FEXCore::X86State::RFLAG_SF_RAW_LOC,
FEXCore::X86State::RFLAG_TF_LOC,
FEXCore::X86State::RFLAG_IF_LOC,
FEXCore::X86State::RFLAG_DF_RAW_LOC,
FEXCore::X86State::RFLAG_DF_LOC,
FEXCore::X86State::RFLAG_OF_RAW_LOC,
FEXCore::X86State::RFLAG_IOPL_LOC,
FEXCore::X86State::RFLAG_NT_LOC,
@@ -38,10 +39,61 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
FEXCore::X86State::RFLAG_ID_LOC,
};
void OpDispatchBuilder::ZeroPF_AF() {
void OpDispatchBuilder::ZeroMultipleFlags(uint32_t FlagsMask) {
auto ZeroConst = _Constant(0);
if (ContainsNZCV(FlagsMask)) {
// NZCV is stored packed together.
// It's more optimal to zero NZCV with move+bic instead of multiple bics.
auto NZCVFlagsMask = FlagsMask & FullNZCVMask;
if (NZCVFlagsMask == FullNZCVMask) {
ZeroNZCV();
}
else {
const auto IndexMask = NZCVIndexMask(FlagsMask);
if (std::popcount(NZCVFlagsMask) == 1) {
// It's more optimal to store only one here.
for (size_t i = 0; NZCVFlagsMask && i < FlagOffsets.size(); ++i) {
const auto FlagOffset = FlagOffsets[i];
const auto FlagMask = 1U << FlagOffset;
if (!(FlagMask & NZCVFlagsMask)) {
continue;
}
SetRFLAG(ZeroConst, FlagOffset);
NZCVFlagsMask &= ~(FlagMask);
}
}
else {
auto IndexMaskConstant = _Constant(IndexMask);
auto NewNZCV = _Andn(OpSize::i64Bit, GetNZCV(), IndexMaskConstant);
SetNZCV(NewNZCV);
}
// Unset the possibly set bits.
PossiblySetNZCVBits &= ~IndexMask;
}
// Handled NZCV, so remove it from the mask.
FlagsMask &= ~FullNZCVMask;
}
// PF is stored inverted, so invert it when we zero.
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Constant(1));
SetAF(0);
if (FlagsMask & (1u << X86State::RFLAG_PF_RAW_LOC)) {
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Constant(1));
FlagsMask &= ~(1u << X86State::RFLAG_PF_RAW_LOC);
}
// Handle remaining masks.
for (size_t i = 0; FlagsMask && i < FlagOffsets.size(); ++i) {
const auto FlagOffset = FlagOffsets[i];
const auto FlagMask = 1U << FlagOffset;
if (!(FlagMask & FlagsMask)) {
continue;
}
SetRFLAG(ZeroConst, FlagOffset);
FlagsMask &= ~(FlagMask);
}
}
void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
@@ -128,7 +180,7 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
// instead.
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_RAW_LOC)) {
// Set every bit except the bottommost.
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false), _Constant(~1ull));
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(), _Constant(~1ull));
// Rotate the bottom bit to the appropriate location for PF, so we get
// something like 111P1111. Then invert that to get 000p0000. Then OR that
@@ -186,21 +238,18 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNo
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
}
OrderedNode *OpDispatchBuilder::LoadPFRaw(bool Invert) {
OrderedNode *OpDispatchBuilder::LoadPFRaw() {
// Read the stored byte. This is the original result (up to 64-bits), it needs
// parity calculated.
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
// Cascade to calculate parity of bottom 8-bits to bottom bit.
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 4);
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 2);
// Cast the input to a 32-bit FPR. Logically we only need 8-bit, but that would
// generate unwanted an ubfx instruction. VPopcount will ignore the upper bits anyway.
auto InputFPR = _VCastFromGPR(4, 4, Result);
if (Invert)
Result = _XornShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
else
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
return Result;
// Calculate the popcount.
auto Count = _VPopcount(1, 1, InputFPR);
return _VExtractToGPR(8, 1, Count, 0);
}
OrderedNode *OpDispatchBuilder::LoadAF() {
@@ -228,43 +277,24 @@ void OpDispatchBuilder::FixupAF() {
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
}
void OpDispatchBuilder::SetAFAndFixup(OrderedNode *AF) {
// We have a value of AF, we shift into AF[4]. We need to fixup AF[4] so that
// we get the right value when we XOR in PF[4] later. The easiest solution is
// to XOR by PF[4], since:
//
// (AF[4] ^ PF[4]) ^ PF[4] = AF[4]
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
OrderedNode *XorRes = _XorShift(OpSize::i32Bit, PFRaw, AF, ShiftType::LSL, 4);
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
}
void OpDispatchBuilder::CalculatePF(OrderedNode *Res) {
// Calculation is entirely deferred until load, just store the 8-bit result.
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Res);
}
void OpDispatchBuilder::CalculateAF(OrderedNode *Src1, OrderedNode *Src2) {
void OpDispatchBuilder::CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
// We only care about bit 4 in the subsequent XOR. If we'll XOR with 0,
// there's no sense XOR'ing at all. If we'll XOR with 1, that's just
// inverting.
// there's no sense XOR'ing at all. This affects INC.
uint64_t Const;
if (IsValueConstant(WrapNode(Src2), &Const)) {
if (Const & (1u << 4)) {
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Not(OpSize::i32Bit, Src1));
} else {
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Src1);
}
if (IsValueConstant(WrapNode(Src2), &Const) && (Const & (1u << 4)) == 0) {
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Src1);
return;
}
// We store the XOR of the arguments. At read time, we XOR with the
// appropriate bit of the result (available as the PF flag) and extract the
// appropriate bit.
OrderedNode *XorRes = _Xor(OpSize::i32Bit, Src1, Src2);
OrderedNode *XorRes = _Xor(OpSize, Src1, Src2);
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
}
@@ -280,9 +310,34 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
}
switch (CurrentDeferredFlags.Type) {
case FlagsGenerationType::TYPE_ADC:
CalculateFlags_ADC(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.ThreeSource.Src1,
CurrentDeferredFlags.Sources.ThreeSource.Src2,
CurrentDeferredFlags.Sources.ThreeSource.Src3);
break;
case FlagsGenerationType::TYPE_SBB:
CalculateFlags_SBB(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.ThreeSource.Src1,
CurrentDeferredFlags.Sources.ThreeSource.Src2,
CurrentDeferredFlags.Sources.ThreeSource.Src3);
break;
case FlagsGenerationType::TYPE_SUB:
CalculateFlags_SUB(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src1,
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src2,
CurrentDeferredFlags.Sources.TwoSrcImmediate.UpdateCF);
break;
case FlagsGenerationType::TYPE_ADD:
CalculateFlags_ADD(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src1,
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src2,
CurrentDeferredFlags.Sources.TwoSrcImmediate.UpdateCF);
@@ -352,6 +407,34 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
break;
case FlagsGenerationType::TYPE_ROR:
CalculateFlags_RotateRight(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.TwoSource.Src1,
CurrentDeferredFlags.Sources.TwoSource.Src2);
break;
case FlagsGenerationType::TYPE_RORI:
CalculateFlags_RotateRightImmediate(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
break;
case FlagsGenerationType::TYPE_ROL:
CalculateFlags_RotateLeft(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.TwoSource.Src1,
CurrentDeferredFlags.Sources.TwoSource.Src2);
break;
case FlagsGenerationType::TYPE_ROLI:
CalculateFlags_RotateLeftImmediate(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
break;
case FlagsGenerationType::TYPE_BEXTR:
CalculateFlags_BEXTR(CurrentDeferredFlags.Res);
break;
@@ -361,10 +444,7 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
CurrentDeferredFlags.Res);
break;
case FlagsGenerationType::TYPE_BLSMSK:
CalculateFlags_BLSMSK(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.OneSource.Src1);
CalculateFlags_BLSMSK(CurrentDeferredFlags.Res);
break;
case FlagsGenerationType::TYPE_BLSR:
CalculateFlags_BLSR(
@@ -381,8 +461,11 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
CurrentDeferredFlags.Res,
CurrentDeferredFlags.Sources.OneSource.Src1);
break;
case FlagsGenerationType::TYPE_ZCNT:
CalculateFlags_ZCNT(
case FlagsGenerationType::TYPE_TZCNT:
CalculateFlags_TZCNT(CurrentDeferredFlags.Res);
break;
case FlagsGenerationType::TYPE_LZCNT:
CalculateFlags_LZCNT(
CurrentDeferredFlags.SrcSize,
CurrentDeferredFlags.Res);
break;
@@ -403,126 +486,150 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
NZCVDirty = false;
}
OrderedNode *OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2) {
void OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
auto Zero = _Constant(0);
auto One = _Constant(1);
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
OrderedNode *Res;
CalculateAF(Src1, Src2);
CalculateAF(OpSize, Res, Src1, Src2);
CalculatePF(Res);
if (SrcSize >= 4) {
HandleNZCV_RMW();
Res = _AdcWithFlags(OpSize, Src1, Src2);
if (NZCVDirty && CachedNZCV)
_StoreNZCV(CachedNZCV);
CachedNZCV = nullptr;
_AdcNZCV(OpSize, Src1, Src2);
PossiblySetNZCVBits = ~0;
} else {
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
Res = _Adc(OpSize, Src1, Src2);
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
auto SelectOpLT = _Select(FEXCore::IR::COND_ULT, Res, Src2, One, Zero);
auto SelectOpLE = _Select(FEXCore::IR::COND_ULE, Res, Src2, One, Zero);
auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, One, SelectOpLE, SelectOpLT);
// SF/ZF
SetNZ_ZeroCV(SrcSize, Res);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
// CF
// Unsigned
{
auto SelectOpLT = _Select(FEXCore::IR::COND_ULT, Res, Src2, One, Zero);
auto SelectOpLE = _Select(FEXCore::IR::COND_ULE, Res, Src2, One, Zero);
auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, One, SelectOpLE, SelectOpLT);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
}
// Signed
CalculateOF(SrcSize, Res, Src1, Src2, false);
}
CalculatePF(Res);
return Res;
}
OrderedNode *OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2) {
void OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
auto Zero = _Constant(0);
auto One = _Constant(1);
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
CalculateAF(Src1, Src2);
CalculateAF(OpSize, Res, Src1, Src2);
CalculatePF(Res);
OrderedNode *Res;
if (SrcSize >= 4) {
// Rectify input carry
CarryInvert();
HandleNZCV_RMW();
Res = _SbbWithFlags(OpSize, Src1, Src2);
if (NZCVDirty && CachedNZCV)
_StoreNZCV(CachedNZCV);
CachedNZCV = nullptr;
NZCVDirty = false;
_SbbNZCV(OpSize, Src1, Src2);
PossiblySetNZCVBits = ~0;
// Rectify output carry
CarryInvert();
} else {
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
Res = _Sub(OpSize, Src1, _Add(OpSize, Src2, CF));
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
auto SelectOpLT = _Select(FEXCore::IR::COND_UGT, Res, Src1, One, Zero);
auto SelectOpLE = _Select(FEXCore::IR::COND_UGE, Res, Src1, One, Zero);
auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, One, SelectOpLE, SelectOpLT);
// SF/ZF
SetNZ_ZeroCV(SrcSize, Res);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
// CF
// Unsigned
{
auto SelectOpLT = _Select(FEXCore::IR::COND_UGT, Res, Src1, One, Zero);
auto SelectOpLE = _Select(FEXCore::IR::COND_UGE, Res, Src1, One, Zero);
auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, One, SelectOpLE, SelectOpLT);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
}
// Signed
CalculateOF(SrcSize, Res, Src1, Src2, true);
}
CalculatePF(Res);
return Res;
}
OrderedNode *OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
void OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
CalculateAF(OpSize, Res, Src1, Src2);
CalculatePF(Res);
// Stash CF before stomping over it
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
HandleNZCVWrite();
CalculateAF(Src1, Src2);
OrderedNode *Res;
// TODO: Could do this path for small sources if we have FEAT_FlagM
if (SrcSize >= 4) {
Res = _SubWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
_SubNZCV(OpSize, Src1, Src2);
CachedNZCV = nullptr;
NZCVDirty = false;
PossiblySetNZCVBits = ~0;
// We only bother inverting CF if we're actually going to update CF.
if (UpdateCF)
CarryInvert();
} else {
_SubNZCV(IR::SizeToOpSize(SrcSize), Src1, Src2);
Res = _Sub(OpSize::i32Bit, Src1, Src2);
// SF/ZF
SetNZ_ZeroCV(SrcSize, Res);
// CF
if (UpdateCF) {
// Grab carry bit from unmasked output.
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SrcSize * 8, true);
}
CalculateOF(SrcSize, Res, Src1, Src2, true);
}
CalculatePF(Res);
// If we're updating CF, we need to invert it for correctness. If we're not
// updating CF, we need to restore the CF since we stomped over it.
if (UpdateCF)
CarryInvert();
else
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
return Res;
}
OrderedNode *OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
// Stash CF before stomping over it
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
HandleNZCVWrite();
CalculateAF(Src1, Src2);
OrderedNode *Res;
if (SrcSize >= 4) {
Res = _AddWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
} else {
_AddNZCV(IR::SizeToOpSize(SrcSize), Src1, Src2);
Res = _Add(OpSize::i32Bit, Src1, Src2);
}
CalculatePF(Res);
// We stomped over CF while calculation flags, restore it.
if (!UpdateCF)
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
}
return Res;
void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
CalculateAF(OpSize, Res, Src1, Src2);
CalculatePF(Res);
// Stash CF before stomping over it
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
// TODO: Could do this path for small sources if we have FEAT_FlagM
if (SrcSize >= 4) {
_AddNZCV(OpSize, Src1, Src2);
CachedNZCV = nullptr;
NZCVDirty = false;
PossiblySetNZCVBits = ~0;
} else {
// SF/ZF
SetNZ_ZeroCV(SrcSize, Res);
// CF
if (UpdateCF) {
// Grab carry bit from unmasked output
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SrcSize * 8, true);
}
CalculateOF(SrcSize, Res, Src1, Src2, false);
}
// We stomped over CF while calculation flags, restore it.
if (!UpdateCF)
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
}
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High) {
HandleNZCVWrite();
// PF/AF/ZF/SF
// Undefined
{
@@ -541,12 +648,13 @@ void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, Or
// undefined, this does what we need.
auto Zero = _Constant(0);
_CondAddNZCV(OpSize::i64Bit, Zero, Zero, CondClassType{COND_EQ}, 0x3 /* nzCV */);
CachedNZCV = nullptr;
NZCVDirty = false;
PossiblySetNZCVBits = ~0;
}
}
void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
HandleNZCVWrite();
auto Zero = _Constant(0);
OpSize Size = IR::SizeToOpSize(GetOpSize(High));
@@ -566,6 +674,9 @@ void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
// If High = 0, then sets to nZcv. Else sets to nzCV. Since SF/ZF undefined,
// this does what we need.
_CondAddNZCV(Size, Zero, Zero, CondClassType{COND_EQ}, 0x3 /* nzCV */);
CachedNZCV = nullptr;
NZCVDirty = false;
PossiblySetNZCVBits = ~0;
}
}
@@ -706,8 +817,8 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
}
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
// Set SF and PF. Clobbers OF, but OF only defined for Shift = 1 where it is
// set below.
// Stash OF before overwriting it
auto OldOF = Shift != 1 ? GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC) : NULL;
SetNZ_ZeroCV(SrcSize, Res);
// CF
@@ -721,6 +832,11 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
// AF
// Undefined
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
// Preserve OF if it won't be written
if (Shift != 1) {
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OldOF);
}
}
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
@@ -758,66 +874,214 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
}
}
void OpDispatchBuilder::CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
auto SizeBits = SrcSize * 8;
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
// OF is set to the XOR of the new CF bit and the most significant bit of the result
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, true);
});
}
void OpDispatchBuilder::CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
// Extract the last bit shifted in to CF
//auto Size = _Constant(GetSrcSize(Res) * 8);
//auto ShiftAmt = _Sub(OpSize::i64Bit, Size, Src2);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
// OF is the LSB and MSB XOR'd together.
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
});
}
void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
if (Shift == 0) return;
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
}
// OF
{
if (Shift == 1) {
// OF is the top two MSBs XOR'd together
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, 1);
}
}
}
void OpDispatchBuilder::CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
if (Shift == 0) return;
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
auto SizeBits = SrcSize * 8;
// Ends up faster overall if we don't have FlagM, slower if we do...
// If Shift != 1, OF is undefined so we choose to zero here.
if (!CTX->HostFeatures.SupportsFlagM)
ZeroCV();
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
}
// OF
{
if (Shift == 1) {
// OF is the LSB and MSB XOR'd together.
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
}
}
}
void OpDispatchBuilder::CalculateFlags_BEXTR(OrderedNode *Src) {
// ZF is set properly. CF and OF are defined as being set to zero. SF, PF, and
// AF are undefined.
SetNZ_ZeroCV(GetOpSize(Src), Src);
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
(1UL << X86State::RFLAG_AF_RAW_LOC));
}
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Result) {
// CF is cleared if Src is zero, otherwise it's set. However, Src is zero iff
// Result is zero, so we can test the result instead. So, CF is just the
// inverted ZF.
//
// ZF/SF/OF set as usual.
SetNZ_ZeroCV(SrcSize, Result);
auto CFOp = GetRFLAG(X86State::RFLAG_ZF_RAW_LOC, true /* Invert */);
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
// PF/AF undefined
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
(1UL << X86State::RFLAG_AF_RAW_LOC));
}
void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
// PF/AF undefined
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
(1UL << X86State::RFLAG_AF_RAW_LOC));
// CF set according to the Src
auto Zero = _Constant(0);
auto One = _Constant(1);
auto CFOp = _Select(IR::COND_EQ, Src, Zero, One, Zero);
// The output of BLSMSK is always nonzero, so TST will clear Z (along with C
// and O) while setting S.
SetNZ_ZeroCV(SrcSize, Result);
// Handle flag setting.
//
// All that matters primarily for this instruction is
// that we only set the ZF flag properly.
//
// CF and OF are defined as being set to zero
//
// Every other flag is considered undefined after a
// BEXTR instruction, but we opt to reliably clear them.
//
ZeroMultipleFlags(FullNZCVMask);
// PF/AF undefined
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
(1UL << X86State::RFLAG_AF_RAW_LOC));
// ZF
auto ZeroOp = _Select(IR::COND_EQ,
Src, Zero,
One, Zero);
SetRFLAG<X86State::RFLAG_ZF_RAW_LOC>(ZeroOp);
}
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src) {
// Now for the flags:
//
// Only CF, SF, ZF and OF are defined as being updated
// CF is cleared if Src is zero, otherwise it's set.
// SF is set to the value of the most significant operand bit of Result.
// OF is always cleared
// ZF is set, as usual, if Result is zero or not.
auto Zero = _Constant(0);
auto One = _Constant(1);
SetNZ_ZeroCV(SrcSize, Src);
// PF/AF undefined
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
(1UL << X86State::RFLAG_AF_RAW_LOC));
// CF
{
auto CFOp = _Select(IR::COND_NEQ, Src, Zero, One, Zero);
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
}
}
void OpDispatchBuilder::CalculateFlags_BLSMSK(OrderedNode *Src) {
// Now for the flags.
auto Zero = _Constant(0);
auto One = _Constant(1);
uint32_t FlagsMaskToZero =
(1U << X86State::RFLAG_ZF_RAW_LOC) |
(1U << X86State::RFLAG_OF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
// PF/AF undefined
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
(1UL << X86State::RFLAG_AF_RAW_LOC));
auto CFOp = _Select(IR::COND_NEQ, Src, Zero, One, Zero);
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
}
void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
// Now for flags.
auto Zero = _Constant(0);
auto One = _Constant(1);
auto CFOp = _Select(IR::COND_EQ, Src, Zero, One, Zero);
SetNZ_ZeroCV(SrcSize, Result);
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
// PF/AF undefined
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
(1UL << X86State::RFLAG_AF_RAW_LOC));
// CF
{
auto CFOp = _Select(IR::COND_NEQ, Src, Zero, One, Zero);
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
}
}
void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode *Result) {
// We need to set ZF while clearing the rest of NZCV. The result of a popcount
// is in the range [0, 63]. In particular, it is always positive. So a
// combined NZ test will correctly zero SF/CF/OF while setting ZF.
SetNZ_ZeroCV(OpSize::i32Bit, Result);
ZeroPF_AF();
void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode *Src) {
// Set ZF
auto Zero = _Constant(0);
auto ZFResult = _Select(FEXCore::IR::COND_EQ,
Src, Zero,
_Constant(1), Zero);
// Set flags
uint32_t FlagsMaskToZero =
FullNZCVMask |
(1U << X86State::RFLAG_AF_RAW_LOC) |
(1U << X86State::RFLAG_PF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(ZFResult);
}
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
@@ -829,24 +1093,45 @@ void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(Src);
}
void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, OrderedNode *Result) {
void OpDispatchBuilder::CalculateFlags_TZCNT(OrderedNode *Src) {
// OF, SF, AF, PF all undefined
// Test ZF of result, SF is undefined so this is ok.
SetNZ_ZeroCV(SrcSize, Result);
ZeroNZCV();
// Now set CF if the Result = SrcSize * 8. Since SrcSize is a power-of-two and
// Result is <= SrcSize * 8, we equivalently check if the log2(SrcSize * 8)
// bit is set. No masking is needed because no higher bits could be set.
unsigned CarryBit = FEXCore::ilog2(SrcSize * 8u);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Result, CarryBit);
auto Zero = _Constant(0);
auto ZFResult = _Select(FEXCore::IR::COND_EQ,
Src, Zero,
_Constant(1), Zero);
// Set flags
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(ZFResult);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Src, 0, true);
}
void OpDispatchBuilder::CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src) {
// OF, SF, AF, PF all undefined
ZeroNZCV();
auto Zero = _Constant(0);
auto ZFResult = _Select(FEXCore::IR::COND_EQ,
Src, Zero,
_Constant(1), Zero);
// Set flags
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(ZFResult);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Src, SrcSize * 8 - 1, true);
}
void OpDispatchBuilder::CalculateFlags_RDRAND(OrderedNode *Src) {
// OF, SF, ZF, AF, PF all zero
ZeroNZCV();
ZeroPF_AF();
// CF is set to the incoming source
uint32_t FlagsMaskToZero =
FullNZCVMask |
(1U << X86State::RFLAG_AF_RAW_LOC) |
(1U << X86State::RFLAG_PF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src);
}
@@ -13,6 +13,7 @@ $end_info$
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
@@ -1103,7 +1104,7 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
const auto ExtractSize = Is256Bit ? 4 : 2;
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *VMask = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_MOVMASKB);
OrderedNode *VMask = _VDupFromGPR(SrcSize, 8, _Constant(0x80'40'20'10'08'04'02'01ULL));
auto VCMP = _VCMPLTZ(SrcSize, 1, Src);
auto VAnd = _VAnd(SrcSize, 1, VCMP, VMask);
@@ -3000,15 +3001,16 @@ void OpDispatchBuilder::XSaveOp(OpcodeArgs) {
XSaveOpImpl(Op);
}
OrderedNode *OpDispatchBuilder::XSaveBase(X86Tables::DecodedOp Op) {
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
return AppendSegmentOffset(Mem, Op->Flags);
}
void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
const auto XSaveBase = [this, Op] {
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
return AppendSegmentOffset(Mem, Op->Flags);
};
// NOTE: Mask should be EAX and EDX concatenated, but we only need to test
// for features that are in the lower 32 bits, so EAX only is sufficient.
OrderedNode *Mask = LoadGPRRegister(X86State::REG_RAX);
OrderedNode *Base = XSaveBase();
const auto OpSize = IR::SizeToOpSize(CTX->GetGPRSize());
const auto StoreIfFlagSet = [&](uint32_t BitIndex, auto fn, uint32_t FieldSize = 1){
@@ -3032,26 +3034,25 @@ void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
// x87
{
StoreIfFlagSet(0, [this, Op] { SaveX87State(Op, XSaveBase(Op)); });
StoreIfFlagSet(0, [this, Op, Base] { SaveX87State(Op, Base); });
}
// SSE
{
StoreIfFlagSet(1, [this, Op] { SaveSSEState(XSaveBase(Op)); });
StoreIfFlagSet(1, [this, Base] { SaveSSEState(Base); });
}
// AVX
if (CTX->HostFeatures.SupportsAVX)
{
StoreIfFlagSet(2, [this, Op] { SaveAVXState(XSaveBase(Op)); });
StoreIfFlagSet(2, [this, Base] { SaveAVXState(Base); });
}
// We need to save MXCSR and MXCSR_MASK if either SSE or AVX are requested to be saved
{
StoreIfFlagSet(1, [this, Op] { SaveMXCSRState(XSaveBase(Op)); }, 2);
StoreIfFlagSet(1, [this, Base] { SaveMXCSRState(Base); }, 2);
}
// Update XSTATE_BV region of the XSAVE header
{
OrderedNode *Base = XSaveBase(Op);
OrderedNode *HeaderOffset = _Add(OpSize, Base, _Constant(512));
// NOTE: We currently only support the first 3 bits (x87, SSE, and AVX)
@@ -3209,11 +3210,14 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
const auto OpSize = IR::SizeToOpSize(CTX->GetGPRSize());
const auto XSaveBase = [this, Op] {
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
return AppendSegmentOffset(Mem, Op->Flags);
};
// Set up base address for the XSAVE region to restore from, and also read the
// XSTATE_BV bit flags out of the XSTATE header.
//
// Note: we rematerialize Base in each block to avoid crossblock liveness.
OrderedNode *Base = XSaveBase(Op);
OrderedNode *Base = XSaveBase();
OrderedNode *Mask = _LoadMem(GPRClass, 8, _Add(OpSize, Base, _Constant(512)), 8);
// If a bit in our XSTATE_BV is set, then we restore from that region of the XSAVE area,
@@ -3249,28 +3253,27 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
// x87
{
RestoreIfFlagSetOrDefault(0,
[this, Op] { RestoreX87State(XSaveBase(Op)); },
[this, Base] { RestoreX87State(Base); },
[this, Op] { DefaultX87State(Op); });
}
// SSE
{
RestoreIfFlagSetOrDefault(1,
[this, Op] { RestoreSSEState(XSaveBase(Op)); },
[this, Base] { RestoreSSEState(Base); },
[this] { DefaultSSEState(); });
}
// AVX
if (CTX->HostFeatures.SupportsAVX)
{
RestoreIfFlagSetOrDefault(2,
[this, Op] { RestoreAVXState(XSaveBase(Op)); },
[this, Base] { RestoreAVXState(Base); },
[this] { DefaultAVXState(); });
}
{
// We need to restore the MXCSR if either SSE or AVX are requested to be saved
RestoreIfFlagSetOrDefault(1,
[this, Op, OpSize] {
OrderedNode *Base = XSaveBase(Op);
[this, Base, OpSize] {
OrderedNode *MXCSRLocation = _Add(OpSize, Base, _Constant(24));
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
RestoreMXCSRState(MXCSR);
@@ -3419,12 +3422,15 @@ void OpDispatchBuilder::VPALIGNROp(OpcodeArgs) {
template<size_t ElementSize>
void OpDispatchBuilder::UCOMISxOp(OpcodeArgs) {
InvalidateDeferredFlags();
const auto SrcSize = Op->Src[0].IsGPR() ? GetGuestVectorLength() : GetSrcSize(Op);
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetGuestVectorLength(), Op->Flags);
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
HandleNZCVWrite();
CachedNZCV = nullptr;
_FCmp(ElementSize, Src1, Src2);
PossiblySetNZCVBits = ~0;
ConvertNZCVToSSE();
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so PF[4] is
@@ -4574,9 +4580,13 @@ void OpDispatchBuilder::PTestOp(OpcodeArgs) {
OrderedNode *Test1 = _VAnd(Size, 1, Dest, Src);
OrderedNode *Test2 = _VBic(Size, 1, Src, Dest);
// Element size must be less than 32-bit for the sign bit tricks.
Test1 = _VUMaxV(Size, 2, Test1);
Test2 = _VUMaxV(Size, 2, Test2);
Test1 = _VPopcount(Size, 1, Test1);
Test2 = _VPopcount(Size, 1, Test2);
// Element size doesn't matter here
// x86-64 doesn't support a horizontal byte add though
Test1 = _VAddV(Size, 2, Test1);
Test2 = _VAddV(Size, 2, Test2);
Test1 = _VExtractToGPR(Size, 2, Test1, 0);
Test2 = _VExtractToGPR(Size, 2, Test2, 0);
@@ -4584,17 +4594,22 @@ void OpDispatchBuilder::PTestOp(OpcodeArgs) {
auto ZeroConst = _Constant(0);
auto OneConst = _Constant(1);
Test1 = _Select(FEXCore::IR::COND_EQ,
Test1, ZeroConst, OneConst, ZeroConst);
Test2 = _Select(FEXCore::IR::COND_EQ,
Test2, ZeroConst, OneConst, ZeroConst);
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
// Set ZF according to Test1. SF will be zeroed since we do a 32-bit test on
// the results of a 16-bit value from the UMaxV, so the 32-bit sign bit is
// cleared even if the 16-bit scalars were negative.
SetNZ_ZeroCV(32, Test1);
ZeroNZCV();
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Test1);
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Test2);
ZeroPF_AF();
uint32_t FlagsMaskToZero =
(1U << X86State::RFLAG_PF_RAW_LOC) |
(1U << X86State::RFLAG_AF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
}
void OpDispatchBuilder::VTESTOpImpl(OpcodeArgs, size_t ElementSize) {
@@ -4615,23 +4630,33 @@ void OpDispatchBuilder::VTESTOpImpl(OpcodeArgs, size_t ElementSize) {
OrderedNode *MaskedAnd = _VAnd(SrcSize, 1, AndTest, Mask);
OrderedNode *MaskedAndNot = _VAnd(SrcSize, 1, AndNotTest, Mask);
OrderedNode *MaxAnd = _VUMaxV(SrcSize, 2, MaskedAnd);
OrderedNode *MaxAndNot = _VUMaxV(SrcSize, 2, MaskedAndNot);
OrderedNode *AndPopCount = _VPopcount(SrcSize, 1, MaskedAnd);
OrderedNode *AndNotPopCount = _VPopcount(SrcSize, 1, MaskedAndNot);
OrderedNode *AndGPR = _VExtractToGPR(SrcSize, 2, MaxAnd, 0);
OrderedNode *AndNotGPR = _VExtractToGPR(SrcSize, 2, MaxAndNot, 0);
OrderedNode *SummedAnd = _VAddV(SrcSize, 2, AndPopCount);
OrderedNode *SummedAndNot = _VAddV(SrcSize, 2, AndNotPopCount);
OrderedNode *AndGPR = _VExtractToGPR(SrcSize, 2, SummedAnd, 0);
OrderedNode *AndNotGPR = _VExtractToGPR(SrcSize, 2, SummedAndNot, 0);
OrderedNode *ZeroConst = _Constant(0);
OrderedNode *OneConst = _Constant(1);
OrderedNode *ZFResult = _Select(IR::COND_EQ, AndGPR, ZeroConst,
OneConst, ZeroConst);
OrderedNode *CFResult = _Select(IR::COND_EQ, AndNotGPR, ZeroConst,
OneConst, ZeroConst);
// As in PTest, this sets Z appropriately while zeroing the rest of NZCV.
SetNZ_ZeroCV(32, AndGPR);
SetRFLAG<X86State::RFLAG_ZF_RAW_LOC>(ZFResult);
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFResult);
ZeroPF_AF();
uint32_t FlagsMaskToZero =
(1U << X86State::RFLAG_PF_RAW_LOC) |
(1U << X86State::RFLAG_AF_RAW_LOC) |
(1U << X86State::RFLAG_SF_RAW_LOC) |
(1U << X86State::RFLAG_OF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
}
template <size_t ElementSize>
@@ -5563,7 +5588,11 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(GetFlagBit(18));
SetRFLAG<X86State::RFLAG_OF_RAW_LOC>(GetFlagBit(19));
ZeroPF_AF();
uint32_t FlagsMaskToZero =
(1U << X86State::RFLAG_PF_RAW_LOC) |
(1U << X86State::RFLAG_AF_RAW_LOC);
ZeroMultipleFlags(FlagsMaskToZero);
}
void OpDispatchBuilder::VPCMPESTRIOp(OpcodeArgs) {
@@ -14,6 +14,7 @@ $end_info$
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/FPState.h>
#include <FEXCore/IR/IREmitter.h>
#include <stddef.h>
#include <stdint.h>
@@ -13,6 +13,7 @@ $end_info$
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/IR/IREmitter.h>
#include <stddef.h>
#include <stdint.h>
@@ -12,16 +12,51 @@ $end_info$
namespace FEXCore::X86Tables {
std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps{};
std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps{};
std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps{};
std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps{};
std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> OpSizeModOps{};
std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps{};
std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps{};
std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps{};
std::array<X86InstInfo, MAX_X87_TABLE_SIZE> X87Ops{};
std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps{};
std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps{};
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps{};
std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps{};
std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps{};
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps{};
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps{};
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps{};
void InitializeBaseTables(Context::OperatingMode Mode);
void InitializeSecondaryTables(Context::OperatingMode Mode);
void InitializePrimaryGroupTables(Context::OperatingMode Mode);
void InitializeSecondaryGroupTables();
void InitializeSecondaryModRMTables();
void InitializeX87Tables();
void InitializeDDDTables();
void InitializeH0F38Tables();
void InitializeH0F3ATables(Context::OperatingMode Mode);
void InitializeVEXTables();
void InitializeXOPTables();
void InitializeEVEXTables();
void InitializeInfoTables(Context::OperatingMode Mode) {
InitializeBaseTables(Mode);
InitializeSecondaryTables(Mode);
InitializePrimaryGroupTables(Mode);
InitializeSecondaryGroupTables();
InitializeSecondaryModRMTables();
InitializeX87Tables();
InitializeDDDTables();
InitializeH0F38Tables();
InitializeH0F3ATables(Mode);
InitializeVEXTables();
InitializeXOPTables();
InitializeEVEXTables();
}
}
@@ -14,10 +14,8 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> Table{};
constexpr U8U8InfoStruct BaseOpTable[] = {
void InitializeBaseTables(Context::OperatingMode Mode) {
static constexpr U8U8InfoStruct BaseOpTable[] = {
// Prefixes
// Operand size overide
{0x66, 1, X86InstInfo{"", TYPE_PREFIX, FLAGS_NONE, 0, nullptr}},
@@ -235,12 +233,6 @@ std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
{0xC4, 2, X86InstInfo{"", TYPE_VEX_TABLE_PREFIX, FLAGS_NONE, 0, nullptr}},
};
GenerateTable(&Table.at(0), BaseOpTable, std::size(BaseOpTable));
return Table;
}();
void InitializeBaseTables(Context::OperatingMode Mode) {
static constexpr U8U8InfoStruct BaseOpTable_64[] = {
{0x06, 2, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{0x0E, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
@@ -299,6 +291,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
};
GenerateTable(&BaseOps.at(0), BaseOpTable, std::size(BaseOpTable));
if (Mode == Context::MODE_64BIT) {
GenerateTable(&BaseOps.at(0), BaseOpTable_64, std::size(BaseOpTable_64));
}
@@ -12,9 +12,8 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps = []() consteval {
std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> Table{};
constexpr U8U8InfoStruct DDDNowOpTable[] = {
void InitializeDDDTables() {
static constexpr U8U8InfoStruct DDDNowOpTable[] = {
{0x0C, 1, X86InstInfo{"PI2FW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x0D, 1, X86InstInfo{"PI2FD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x1C, 1, X86InstInfo{"PF2IW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
@@ -53,8 +52,6 @@ std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps = []() consteval {
{0xBF, 1, X86InstInfo{"PAVGUSB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
};
GenerateTable(&Table.at(0), DDDNowOpTable, std::size(DDDNowOpTable));
return Table;
}();
GenerateTable(&DDDNowOps.at(0), DDDNowOpTable, std::size(DDDNowOpTable));
}
}
@@ -11,9 +11,9 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps = []() consteval {
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> Table{};
constexpr U16U8InfoStruct EVEXTable[] = {
void InitializeEVEXTables() {
static constexpr U16U8InfoStruct EVEXTable[] = {
{0x10, 1, X86InstInfo{"VMOVUPS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
{0x11, 1, X86InstInfo{"VMOVUPS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
{0x18, 1, X86InstInfo{"VBROADCASTSS", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
@@ -29,9 +29,6 @@ std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps = []() consteval {
{0xE7, 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
};
GenerateTable(&Table.at(0), EVEXTable, std::size(EVEXTable));
return Table;
}();
GenerateTable(&EVEXTableOps.at(0), EVEXTable, std::size(EVEXTable));
}
}
@@ -12,16 +12,15 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps = []() consteval {
std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> Table{};
void InitializeH0F38Tables() {
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
constexpr uint16_t PF_38_NONE = 0;
constexpr uint16_t PF_38_66 = (1U << 0);
constexpr uint16_t PF_38_F2 = (1U << 1);
constexpr uint16_t PF_38_F3 = (1U << 2);
constexpr U16U8InfoStruct H0F38Table[] = {
static constexpr U16U8InfoStruct H0F38Table[] = {
{OPD(PF_38_NONE, 0x00), 1, X86InstInfo{"PSHUFB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{OPD(PF_38_66, 0x00), 1, X86InstInfo{"PSHUFB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
{OPD(PF_38_NONE, 0x01), 1, X86InstInfo{"PHADDW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
@@ -118,8 +117,6 @@ std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps = []() consteval {
};
#undef OPD
GenerateTable(&Table.at(0), H0F38Table, std::size(H0F38Table));
return Table;
}();
GenerateTable(&H0F38TableOps.at(0), H0F38Table, std::size(H0F38Table));
}
}
@@ -14,13 +14,13 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode)
constexpr uint16_t PF_3A_NONE = 0;
constexpr uint16_t PF_3A_66 = 1;
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps = []() consteval {
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> Table{};
constexpr U16U8InfoStruct H0F3ATable[] = {
void InitializeH0F3ATables(Context::OperatingMode Mode) {
#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode)
constexpr uint16_t PF_3A_NONE = 0;
constexpr uint16_t PF_3A_66 = 1;
static constexpr U16U8InfoStruct H0F3ATable[] = {
{OPD(0, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
{OPD(0, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
{OPD(0, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
@@ -54,11 +54,6 @@ std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps = []() consteval {
{OPD(0, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
};
GenerateTable(&Table.at(0), H0F3ATable, std::size(H0F3ATable));
return Table;
}();
void InitializeH0F3ATables(Context::OperatingMode Mode) {
static constexpr U16U8InfoStruct H0F3ATable_64[] = {
{OPD(1, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
{OPD(1, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRQ", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
@@ -67,6 +62,8 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
#undef OPD
GenerateTable(&H0F3ATableOps.at(0), H0F3ATable, std::size(H0F3ATable));
if (Mode == Context::MODE_64BIT) {
GenerateTable(&H0F3ATableOps.at(0), H0F3ATable_64, std::size(H0F3ATable_64));
}
@@ -13,10 +13,10 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps = []() consteval {
std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> Table{};
void InitializePrimaryGroupTables(Context::OperatingMode Mode) {
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
constexpr U16U8InfoStruct PrimaryGroupOpTable[] = {
const U16U8InfoStruct PrimaryGroupOpTable[] = {
// GROUP_1 | 0x80 | reg
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 0), 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, nullptr}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 1), 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, nullptr}},
@@ -141,13 +141,9 @@ std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps = []() co
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 7), 1, X86InstInfo{"XBEGIN", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT | FLAGS_SETS_RIP | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
};
GenerateTable(&Table.at(0), PrimaryGroupOpTable, std::size(PrimaryGroupOpTable));
return Table;
}();
void InitializePrimaryGroupTables(Context::OperatingMode Mode) {
const U16U8InfoStruct PrimaryGroupOpTable_64[] = {
// Invalid in 64bit mode
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 0), 8, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
@@ -167,6 +163,7 @@ void InitializePrimaryGroupTables(Context::OperatingMode Mode) {
#undef OPD
GenerateTable(&PrimaryInstGroupOps.at(0), PrimaryGroupOpTable, std::size(PrimaryGroupOpTable));
if (Mode == Context::MODE_64BIT) {
GenerateTable(&PrimaryInstGroupOps.at(0), PrimaryGroupOpTable_64, std::size(PrimaryGroupOpTable_64));
}
@@ -12,15 +12,15 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = []() consteval {
std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> Table{};
void InitializeSecondaryGroupTables() {
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
constexpr uint16_t PF_NONE = 0;
constexpr uint16_t PF_F3 = 1;
constexpr uint16_t PF_66 = 2;
constexpr uint16_t PF_F2 = 3;
constexpr U16U8InfoStruct SecondaryExtensionOpTable[] = {
static constexpr U16U8InfoStruct SecondaryExtensionOpTable[] = {
// GROUP 1
// GROUP 2
// GROUP 3
@@ -162,7 +162,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
{OPD(TYPE_GROUP_9, PF_F3, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"RDPID", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"RDPID", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
@@ -487,8 +487,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
};
#undef OPD
GenerateTable(&Table.at(0), SecondaryExtensionOpTable, std::size(SecondaryExtensionOpTable));
return Table;
}();
GenerateTable(&SecondInstGroupOps.at(0), SecondaryExtensionOpTable, std::size(SecondaryExtensionOpTable));
}
}
@@ -11,9 +11,9 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps = []() consteval {
std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> Table{};
constexpr U8U8InfoStruct SecondaryModRMExtensionOpTable[] = {
void InitializeSecondaryModRMTables() {
static constexpr U8U8InfoStruct SecondaryModRMExtensionOpTable[] = {
// REG /1
{((0 << 3) | 0), 1, X86InstInfo{"MONITOR", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
{((0 << 3) | 1), 1, X86InstInfo{"MWAIT", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
@@ -55,8 +55,6 @@ std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps = []()
{((3 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
};
GenerateTable(&Table.at(0), SecondaryModRMExtensionOpTable, std::size(SecondaryModRMExtensionOpTable));
return Table;
}();
GenerateTable(&SecondModRMTableOps.at(0), SecondaryModRMExtensionOpTable, std::size(SecondaryModRMExtensionOpTable));
}
}
@@ -13,10 +13,9 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
auto BaseOpsLambda = []() consteval {
std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> Table{};
constexpr U8U8InfoStruct TwoByteOpTable[] = {
void InitializeSecondaryTables(Context::OperatingMode Mode) {
static constexpr U8U8InfoStruct TwoByteOpTable[] = {
// Instructions
{0x00, 1, X86InstInfo{"", TYPE_GROUP_6, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x01, 1, X86InstInfo{"", TYPE_GROUP_7, FLAGS_NO_OVERLAY, 0, nullptr}},
@@ -167,13 +166,11 @@ auto BaseOpsLambda = []() consteval {
{0x9E, 1, X86InstInfo{"SETLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
{0x9F, 1, X86InstInfo{"SETNLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA0, 2, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA4, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1, nullptr}},
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA6, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA8, 2, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
{0xAA, 1, X86InstInfo{"RSM", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
{0xAB, 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xAC, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1, nullptr}},
@@ -268,16 +265,23 @@ auto BaseOpsLambda = []() consteval {
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0, nullptr}},
};
GenerateTable(&Table.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
static constexpr U8U8InfoStruct TwoByteOpTable_32[] = {
{0xA0, 1, X86InstInfo{"PUSH FS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA1, 1, X86InstInfo{"POP FS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
return Table;
};
{0xA8, 1, X86InstInfo{"PUSH GS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA9, 1, X86InstInfo{"POP GS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
};
std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = BaseOpsLambda();
std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps = []() consteval {
std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> Table{};
static constexpr U8U8InfoStruct TwoByteOpTable_64[] = {
{0xA0, 1, X86InstInfo{"PUSH FS", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA1, 1, X86InstInfo{"POP FS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
constexpr U8U8InfoStruct RepModOpTable[] = {
{0xA8, 1, X86InstInfo{"PUSH GS", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA9, 1, X86InstInfo{"POP GS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
};
static constexpr U8U8InfoStruct RepModOpTable[] = {
{0x0, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
{0x10, 1, X86InstInfo{"MOVSS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
@@ -357,15 +361,7 @@ std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps = []() consteval {
{0xFF, 1, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
};
GenerateTableWithCopy(&Table.at(0), RepModOpTable, std::size(RepModOpTable), &BaseOpsLambda().at(0));
return Table;
}();
std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() consteval {
std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> Table{};
constexpr U8U8InfoStruct RepNEModOpTable[] = {
static constexpr U8U8InfoStruct RepNEModOpTable[] = {
{0x0, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
{0x10, 1, X86InstInfo{"MOVSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
@@ -438,15 +434,7 @@ std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() consteval {
{0xF8, 8, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
};
GenerateTableWithCopy(&Table.at(0), RepNEModOpTable, std::size(RepNEModOpTable), &BaseOpsLambda().at(0));
return Table;
}();
std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> OpSizeModOps = []() consteval {
std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> Table{};
constexpr U8U8InfoStruct OpSizeModOpTable[] = {
static constexpr U8U8InfoStruct OpSizeModOpTable[] = {
{0x0, 16, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
{0x10, 1, X86InstInfo{"MOVUPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
@@ -593,40 +581,19 @@ std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> OpSizeModOps = []() consteval
{0xFF, 1, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
};
GenerateTableWithCopy(&Table.at(0), OpSizeModOpTable, std::size(OpSizeModOpTable), &BaseOpsLambda().at(0));
return Table;
}();
void InitializeSecondaryTables(Context::OperatingMode Mode) {
static constexpr U8U8InfoStruct TwoByteOpTable_32[] = {
{0xA0, 1, X86InstInfo{"PUSH FS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA1, 1, X86InstInfo{"POP FS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA8, 1, X86InstInfo{"PUSH GS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA9, 1, X86InstInfo{"POP GS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
};
static constexpr U8U8InfoStruct TwoByteOpTable_64[] = {
{0xA0, 1, X86InstInfo{"PUSH FS", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA1, 1, X86InstInfo{"POP FS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA8, 1, X86InstInfo{"PUSH GS", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xA9, 1, X86InstInfo{"POP GS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_64BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, nullptr}},
};
GenerateTable(&SecondBaseOps.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
if (Mode == Context::MODE_64BIT) {
LateInitCopyTable(&SecondBaseOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
LateInitCopyTable(&RepModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
LateInitCopyTable(&RepNEModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
LateInitCopyTable(&OpSizeModOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
GenerateTable(&SecondBaseOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
}
else {
LateInitCopyTable(&SecondBaseOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
LateInitCopyTable(&RepModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
LateInitCopyTable(&RepNEModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
LateInitCopyTable(&OpSizeModOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
GenerateTable(&SecondBaseOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
}
GenerateTableWithCopy(&RepModOps.at(0), RepModOpTable, std::size(RepModOpTable), &SecondBaseOps.at(0));
GenerateTableWithCopy(&RepNEModOps.at(0), RepNEModOpTable, std::size(RepNEModOpTable), &SecondBaseOps.at(0));
GenerateTableWithCopy(&OpSizeModOps.at(0), OpSizeModOpTable, std::size(OpSizeModOpTable), &SecondBaseOps.at(0));
}
}
@@ -11,10 +11,10 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> Table{};
void InitializeVEXTables() {
#define OPD(map_select, pp, opcode) (((map_select - 1) << 10) | (pp << 8) | (opcode))
constexpr U16U8InfoStruct VEXTable[] = {
static constexpr U16U8InfoStruct VEXTable[] = {
// Map 0 (Reserved)
// VEX Map 1
{OPD(1, 0b00, 0x10), 1, X86InstInfo{"VMOVUPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
@@ -488,15 +488,8 @@ std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
};
#undef OPD
GenerateTable(&Table.at(0), VEXTable, std::size(VEXTable));
return Table;
}();
std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps = []() consteval {
std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> Table{};
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
constexpr U8U8InfoStruct VEXGroupTable[] = {
static constexpr U8U8InfoStruct VEXGroupTable[] = {
{OPD(TYPE_VEX_GROUP_12, 1, 0b010), 1, X86InstInfo{"VPSRLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
{OPD(TYPE_VEX_GROUP_12, 1, 0b100), 1, X86InstInfo{"VPSRAW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
{OPD(TYPE_VEX_GROUP_12, 1, 0b110), 1, X86InstInfo{"VPSLLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
@@ -519,8 +512,7 @@ std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps = []() conste
};
#undef OPD
GenerateTable(&Table.at(0), VEXGroupTable, std::size(VEXGroupTable));
return Table;
}();
GenerateTable(&VEXTableOps.at(0), VEXTable, std::size(VEXTable));
GenerateTable(&VEXTableGroupOps.at(0), VEXGroupTable, std::size(VEXGroupTable));
}
}
@@ -507,30 +507,26 @@ using U8U8InfoStruct = X86TablesInfoStruct<uint8_t>;
using U16U8InfoStruct = X86TablesInfoStruct<uint16_t>;
template<typename OpcodeType>
constexpr static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
for (size_t j = 0; j < TableSize; ++j) {
X86TablesInfoStruct<OpcodeType> const &Op = LocalTable[j];
auto OpNum = Op.first;
X86InstInfo const &Info = Op.Info;
for (uint32_t i = 0; i < Op.second; ++i) {
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
}
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
FinalTable[OpNum + i] = Info;
}
}
};
template<typename OpcodeType>
constexpr static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize, X86InstInfo *OtherLocal) {
static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize, X86InstInfo *OtherLocal) {
for (size_t j = 0; j < TableSize; ++j) {
X86TablesInfoStruct<OpcodeType> const &Op = LocalTable[j];
auto OpNum = Op.first;
X86InstInfo const &Info = Op.Info;
for (uint32_t i = 0; i < Op.second; ++i) {
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
}
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
if (Info.Type == TYPE_COPY_OTHER) {
FinalTable[OpNum + i] = OtherLocal[OpNum + i];
}
@@ -542,30 +538,13 @@ constexpr static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86T
};
template<typename OpcodeType>
static inline void LateInitCopyTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *OtherLocal, size_t OtherTableSize) {
for (size_t j = 0; j < OtherTableSize; ++j) {
X86TablesInfoStruct<OpcodeType> const &OtherOp = OtherLocal[j];
auto OtherOpNum = OtherOp.first;
X86InstInfo const &OtherInfo = OtherOp.Info;
for (uint32_t i = 0; i < OtherOp.second; ++i) {
X86InstInfo &FinalOp = FinalTable[OtherOpNum + i];
if (FinalOp.Type == TYPE_COPY_OTHER) {
FinalOp = OtherInfo;
}
}
}
}
template<typename OpcodeType>
constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
for (size_t j = 0; j < TableSize; ++j) {
X86TablesInfoStruct<OpcodeType> const &Op = LocalTable[j];
auto OpNum = Op.first;
X86InstInfo const &Info = Op.Info;
for (uint32_t i = 0; i < Op.second; ++i) {
if (FinalTable[OpNum + i].Type != TYPE_UNKNOWN) {
ERROR_AND_DIE_FMT("Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
}
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
if ((OpNum & 0b11'000'000) == 0b11'000'000) {
// If the mod field is 0b11 then it is a regular op
FinalTable[OpNum + i] = Info;
@@ -573,9 +552,7 @@ constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86Tables
else {
// If the mod field is !0b11 then this instruction is duplicated through the whole mod [0b00, 0b10] range
// and the modrm.rm space because that is used part of the instruction encoding
if ((OpNum & 0b11'000'000) != 0) {
ERROR_AND_DIE_FMT("Only support mod field of zero in this path");
}
LOGMAN_THROW_AA_FMT((OpNum & 0b11'000'000) == 0, "Only support mod field of zero in this path");
for (uint16_t mod = 0b00'000'000; mod < 0b11'000'000; mod += 0b01'000'000) {
for (uint16_t rm = 0b000; rm < 0b1'000; ++rm) {
FinalTable[(OpNum | mod | rm) + i] = Info;
@@ -11,11 +11,11 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_X87_TABLE_SIZE> X87Ops = []() consteval {
std::array<X86InstInfo, MAX_X87_TABLE_SIZE> Table{};
void InitializeX87Tables() {
#define OPD(op, modrmop) (((op - 0xD8) << 8) | modrmop)
#define OPDReg(op, reg) (((op - 0xD8) << 8) | (reg << 3))
constexpr U16U8InfoStruct X87OpTable[] = {
static constexpr U16U8InfoStruct X87OpTable[] = {
// 0xD8
{OPDReg(0xD8, 0), 1, X86InstInfo{"FADD", TYPE_X87, FLAGS_MODRM, 0, nullptr}},
{OPDReg(0xD8, 1), 1, X86InstInfo{"FMUL", TYPE_X87, FLAGS_MODRM, 0, nullptr}},
@@ -263,8 +263,6 @@ std::array<X86InstInfo, MAX_X87_TABLE_SIZE> X87Ops = []() consteval {
#undef OPD
#undef OPDReg
GenerateX87Table(&Table.at(0), X87OpTable, std::size(X87OpTable));
return Table;
}();
GenerateX87Table(&X87Ops.at(0), X87OpTable, std::size(X87OpTable));
}
}
@@ -12,14 +12,14 @@ $end_info$
namespace FEXCore::X86Tables {
using namespace InstFlags;
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps = []() consteval {
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> Table{};
void InitializeXOPTables() {
#define OPD(group, pp, opcode) ( (group << 10) | (pp << 8) | (opcode))
constexpr uint16_t XOP_GROUP_8 = 0;
constexpr uint16_t XOP_GROUP_9 = 1;
constexpr uint16_t XOP_GROUP_A = 2;
constexpr U16U8InfoStruct XOPTable[] = {
static constexpr U16U8InfoStruct XOPTable[] = {
// Group 8
{OPD(XOP_GROUP_8, 0, 0x85), 1, X86InstInfo{"VPMAXSSWW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
{OPD(XOP_GROUP_8, 0, 0x86), 1, X86InstInfo{"VPMACSSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
@@ -104,15 +104,8 @@ std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps = []() consteval {
};
#undef OPD
GenerateTable(&Table.at(0), XOPTable, std::size(XOPTable));
return Table;
}();
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps = []() consteval {
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> Table{};
#define OPD(subgroup, opcode) (((subgroup - 1) << 3) | (opcode))
constexpr U8U8InfoStruct XOPGroupTable[] = {
static constexpr U8U8InfoStruct XOPGroupTable[] = {
// Group 1
{OPD(1, 1), 1, X86InstInfo{"BLCFILL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
{OPD(1, 2), 1, X86InstInfo{"BLSFILL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
@@ -136,8 +129,7 @@ std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps = []() conste
};
#undef OPD
GenerateTable(&Table.at(0), XOPGroupTable, std::size(XOPGroupTable));
return Table;
}();
GenerateTable(&XOPTableOps.at(0), XOPTable, std::size(XOPTable));
GenerateTable(&XOPTableGroupOps.at(0), XOPGroupTable, std::size(XOPGroupTable));
}
}
+3 -27
View File
@@ -6,13 +6,12 @@ tags: glue|thunks
$end_info$
*/
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
@@ -181,9 +180,6 @@ namespace FEXCore {
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = (uintptr_t)arg0;
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = (uintptr_t)arg1;
} else {
if ((reinterpret_cast<uintptr_t>(arg1) >> 32) != 0) {
ERROR_AND_DIE_FMT("Tried to call guest function with arguments packed to a 64-bit address");
}
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RCX] = (uintptr_t)arg0;
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDX] = (uintptr_t)arg1;
}
@@ -220,7 +216,7 @@ namespace FEXCore {
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}",
args->original_callee, args->target_addr);
auto Result = CTX->AddCustomIREntrypoint(
auto Result = Thread->CTX->AddCustomIREntrypoint(
args->original_callee,
[CTX, GuestThunkEntrypoint = args->target_addr](uintptr_t Entrypoint, FEXCore::IR::IREmitter *emit) {
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0);
@@ -487,26 +483,6 @@ namespace FEXCore {
}
}
FEX_DEFAULT_VISIBILITY void* GetGuestStack() {
if (!Thread) {
ERROR_AND_DIE_FMT("Thunked library attempted to query guest stack pointer asynchronously");
}
return (void*)(uintptr_t)((Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP]));
}
FEX_DEFAULT_VISIBILITY void MoveGuestStack(uintptr_t NewAddress) {
if (!Thread) {
ERROR_AND_DIE_FMT("Thunked library attempted to query guest stack pointer asynchronously");
}
if (NewAddress >> 32) {
ERROR_AND_DIE_FMT("Tried to set stack pointer for 32-bit guest to a 64-bit address");
}
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP] = NewAddress;
}
#else
fextl::unique_ptr<ThunkHandler> ThunkHandler::Create() {
ERROR_AND_DIE_FMT("Unsupported");
+1 -2
View File
@@ -7,8 +7,7 @@ $end_info$
#pragma once
#include "Interface/IR/IR.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/vector.h>
+2 -2
View File
@@ -2,9 +2,9 @@
#include "FEXHeaderUtils/Filesystem.h"
#include "Interface/Context/Context.h"
#include "Interface/IR/AOTIR.h"
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/fextl/fmt.h>
+1 -2
View File
@@ -1,8 +1,7 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Interface/IR/RegisterAllocationData.h"
#include "FEXCore/IR/RegisterAllocationData.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/string.h>
-636
View File
@@ -1,636 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/ThreadPoolAllocator.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/sstream.h>
namespace FEXCore::IR {
class OrderedNode;
class RegisterAllocationPass;
class RegisterAllocationData;
/**
* @brief The IROp_Header is an dynamically sized array
* At the end it contains a uint8_t for the number of arguments that Op has
* Then there is an unsized array of NodeWrapper arguments for the number of arguments this op has
* The op structures that are including the header must ensure that they pad themselves correctly to the number of arguments used
*/
struct IROp_Header;
/**
* @brief Represents the ID of a given IR node.
*
* Intended to provide strong typing from other integer values
* to prevent passing incorrect values to certain API functions.
*/
struct NodeID final {
using value_type = uint32_t;
constexpr NodeID() noexcept = default;
constexpr explicit NodeID(value_type Value_) noexcept : Value{Value_} {}
constexpr NodeID(const NodeID&) noexcept = default;
constexpr NodeID& operator=(const NodeID&) noexcept = default;
constexpr NodeID(NodeID&&) noexcept = default;
constexpr NodeID& operator=(NodeID&&) noexcept = default;
[[nodiscard]] constexpr bool IsValid() const noexcept {
return Value != 0;
}
[[nodiscard]] constexpr bool IsInvalid() const noexcept {
return !IsValid();
}
constexpr void Invalidate() noexcept {
Value = 0;
}
[[nodiscard]] friend constexpr bool operator==(NodeID, NodeID) noexcept = default;
[[nodiscard]] friend constexpr bool operator<(NodeID lhs, NodeID rhs) noexcept {
return lhs.Value < rhs.Value;
}
[[nodiscard]] friend constexpr bool operator>(NodeID lhs, NodeID rhs) noexcept {
return operator<(rhs, lhs);
}
[[nodiscard]] friend constexpr bool operator<=(NodeID lhs, NodeID rhs) noexcept {
return !operator>(lhs, rhs);
}
[[nodiscard]] friend constexpr bool operator>=(NodeID lhs, NodeID rhs) noexcept {
return !operator<(lhs, rhs);
}
friend std::ostream& operator<<(std::ostream& out, NodeID ID) {
out << ID.Value;
return out;
}
friend std::istream& operator>>(std::istream& in, NodeID& ID) {
in >> ID.Value;
return in;
}
value_type Value{};
};
/**
* @brief This is a very simple wrapper for our node pointers
* You probably don't want to use this directly
* Use OpNodeWrapper and OrderedNodeWrapper types below instead
*
* This is necessary to allow two things
* - Reduce memory usage by having the pointer be an 32bit offset rather than the whole 64bit pointer
* - Actually use an offset from a base so we aren't storing pointers for everything
* - Makes IR list copying be as cheap as a memcpy
* Downsides
* - The IR nodes have to be allocated out of a linear array of memory
* - We currently only allow a 32bit offset, so *only* 4 million nodes per list
* - We have to have the base offset live somewhere else
* - Has to be POD and trivially copyable
* - Makes every real node access turn in to a [Base + Offset] access
* - Can be confusing if you're mixing OpNodeWrapper and OrderedNodeWrapper usage
*/
template<typename Type>
struct NodeWrapperBase final {
// On x86-64 using a uint64_t type is more efficient since RIP addressing gives you [<Base> + <Index> + <imm offset>]
// On AArch64 using uint32_t is just more memory efficient. 32bit or 64bit offset doesn't matter
// We use uint32_t to be more memory efficient (Cuts our node list size in half)
using NodeOffsetType = uint32_t;
NodeOffsetType NodeOffset;
explicit NodeWrapperBase() = default;
[[nodiscard]] static NodeWrapperBase WrapOffset(NodeOffsetType Offset) {
NodeWrapperBase Wrapped;
Wrapped.NodeOffset = Offset;
return Wrapped;
}
[[nodiscard]] static NodeWrapperBase WrapPtr(uintptr_t Base, uintptr_t Value) {
NodeWrapperBase Wrapped;
Wrapped.SetOffset(Base, Value);
return Wrapped;
}
[[nodiscard]] static void *UnwrapNode(uintptr_t Base, NodeWrapperBase Node) {
return Node.GetNode(Base);
}
[[nodiscard]] NodeID ID() const;
[[nodiscard]] bool IsInvalid() const { return NodeOffset == 0; }
[[nodiscard]] Type *GetNode(uintptr_t Base) {
return reinterpret_cast<Type*>(Base + NodeOffset);
}
[[nodiscard]] const Type *GetNode(uintptr_t Base) const {
return reinterpret_cast<const Type*>(Base + NodeOffset);
}
void SetOffset(uintptr_t Base, uintptr_t Value) { NodeOffset = Value - Base; }
[[nodiscard]] friend constexpr bool operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
};
static_assert(std::is_trivial_v<NodeWrapperBase<OrderedNode>>);
static_assert(sizeof(NodeWrapperBase<OrderedNode>) == sizeof(uint32_t));
using OpNodeWrapper = NodeWrapperBase<IROp_Header>;
using OrderedNodeWrapper = NodeWrapperBase<OrderedNode>;
struct OrderedNodeHeader {
OpNodeWrapper Value;
OrderedNodeWrapper Next;
OrderedNodeWrapper Previous;
};
static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3);
/**
* @brief This is a node in our IR representation
* Is a doubly linked list node that lives in a representation of a linearly allocated node list
* The links in the nodes can live in a list independent of the data IR data
*
* ex.
* Region1 : ... <-> <OrderedNode> <-> <OrderedNode> <-> ...
* | *<Value> |
* v v
* Region2 : <IROp>..<IROp>..<IROp>..<IROp>
*
* In this example the OrderedNodes are allocated in one linear memory region (Not necessarily contiguous with one another linking)
* The second region is contiguous but they don't have any relationship with one another directly
*/
class OrderedNode final {
friend class NodeWrapperIterator;
friend class OrderedList;
public:
// These three values are laid out very specifically to make it fast to access the NodeWrappers specifically
OrderedNodeHeader Header;
uint32_t NumUses;
using value_type = OrderedNodeWrapper;
OrderedNode() = default;
/**
* @brief Appends a node to this current node
*
* Before. <Prev> <-> <Current> <-> <Next>
* After. <Prev> <-> <Current> <-> <Node> <-> Next
*
* @return Pointer to the node being added
*/
value_type append(uintptr_t Base, value_type Node) {
// Set Next Node's Previous to incoming node
SetPrevious(Base, Header.Next, Node);
// Set Incoming node's links to this node's links
SetPrevious(Base, Node, Wrapped(Base));
SetNext(Base, Node, Header.Next);
// Set this node's next to the incoming node
SetNext(Base, Wrapped(Base), Node);
// Return the node we are appending
return Node;
}
OrderedNode *append(uintptr_t Base, OrderedNode *Node) {
value_type WNode = Node->Wrapped(Base);
// Set Next Node's Previous to incoming node
SetPrevious(Base, Header.Next, WNode);
// Set Incoming node's links to this node's links
SetPrevious(Base, WNode, Wrapped(Base));
SetNext(Base, WNode, Header.Next);
// Set this node's next to the incoming node
SetNext(Base, Wrapped(Base), WNode);
// Return the node we are appending
return Node;
}
/**
* @brief Prepends a node to the current node
* Before. <Prev> <-> <Current> <-> <Next>
* After. <Prev> <-> <Node> <-> <Current> <-> Next
*
* @return Pointer to the node being added
*/
value_type prepend(uintptr_t Base, value_type Node) {
// Set the previous node's next to the incoming node
SetNext(Base, Header.Previous, Node);
// Set the incoming node's links
SetPrevious(Base, Node, Header.Previous);
SetNext(Base, Node, Wrapped(Base));
// Set the current node's link
SetPrevious(Base, Wrapped(Base), Node);
// Return the node we are prepending
return Node;
}
OrderedNode *prepend(uintptr_t Base, OrderedNode *Node) {
value_type WNode = Node->Wrapped(Base);
// Set the previous node's next to the incoming node
SetNext(Base, Header.Previous, WNode);
// Set the incoming node's links
SetPrevious(Base, WNode, Header.Previous);
SetNext(Base, WNode, Wrapped(Base));
// Set the current node's link
SetPrevious(Base, Wrapped(Base), WNode);
// Return the node we are prepending
return Node;
}
/**
* @brief Gets the remaining size of the blocks from this point onward
*
* Doesn't find the head of the list
*
*/
[[nodiscard]] size_t size(uintptr_t Base) const {
size_t Size = 1;
// Walk the list forward until we hit a sentinel
value_type Current = Header.Next;
while (Current.NodeOffset != 0) {
++Size;
OrderedNode *RealNode = Current.GetNode(Base);
Current = RealNode->Header.Next;
}
return Size;
}
void Unlink(uintptr_t Base) {
// This removes the node from the list. Orphaning it
// Before: <Previous> <-> <Current> <-> <Next>
// After: <Previous <-> <Next>
SetNext(Base, Header.Previous, Header.Next);
SetPrevious(Base, Header.Next, Header.Previous);
}
[[nodiscard]] IROp_Header const* Op(uintptr_t Base) const {
return Header.Value.GetNode(Base);
}
[[nodiscard]] IROp_Header *Op(uintptr_t Base) {
return Header.Value.GetNode(Base);
}
[[nodiscard]] uint32_t GetUses() const { return NumUses; }
void AddUse() { ++NumUses; }
void RemoveUse() { --NumUses; }
[[nodiscard]] value_type Wrapped(uintptr_t Base) const {
value_type Tmp;
Tmp.SetOffset(Base, reinterpret_cast<uintptr_t>(this));
return Tmp;
}
private:
[[nodiscard]] value_type WrappedOffset(uint32_t Offset) const {
value_type Tmp;
Tmp.NodeOffset = Offset;
return Tmp;
}
static void SetPrevious(uintptr_t Base, value_type Node, value_type New) {
OrderedNode *RealNode = Node.GetNode(Base);
RealNode->Header.Previous = New;
}
static void SetNext(uintptr_t Base, value_type Node, value_type New) {
OrderedNode *RealNode = Node.GetNode(Base);
RealNode->Header.Next = New;
}
void SetUses(uint32_t Uses) { NumUses = Uses; }
};
static_assert(std::is_trivial_v<OrderedNode>);
static_assert(std::is_trivially_copyable_v<OrderedNode>);
static_assert(offsetof(OrderedNode, Header) == 0);
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
struct RegisterClassType final {
using value_type = uint32_t;
value_type Val;
[[nodiscard]] constexpr operator value_type() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
};
struct CondClassType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const CondClassType&, const CondClassType&) = default;
};
struct MemOffsetType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const MemOffsetType&, const MemOffsetType&) = default;
};
struct TypeDefinition final {
uint16_t Val;
[[nodiscard]] constexpr operator uint16_t() const {
return Val;
}
[[nodiscard]] static constexpr TypeDefinition Create(uint8_t Bytes) {
TypeDefinition Type{};
Type.Val = Bytes << 8;
return Type;
}
[[nodiscard]] static constexpr TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
TypeDefinition Type{};
Type.Val = (Bytes << 8) | (Elements & 255);
return Type;
}
[[nodiscard]] constexpr uint8_t Bytes() const {
return Val >> 8;
}
[[nodiscard]] constexpr uint8_t Elements() const {
return Val & 255;
}
[[nodiscard]] friend constexpr bool operator==(const TypeDefinition&, const TypeDefinition&) = default;
};
static_assert(std::is_trivial_v<TypeDefinition>);
struct FenceType final {
using value_type = uint8_t;
value_type Val;
[[nodiscard]] constexpr operator value_type() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
};
struct RoundType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]] friend constexpr bool operator==(const RoundType&, const RoundType&) = default;
};
class NodeIterator;
/* This iterator can be used to step though nodes.
* Due to how our IR is laid out, this can be used to either step
* though the CodeBlocks or though the code within a single block.
*/
class NodeIterator {
public:
using value_type = std::tuple<OrderedNode*, IROp_Header*>;
using size_type = std::size_t;
using difference_type = std::ptrdiff_t;
using reference = value_type&;
using const_reference = const value_type&;
using pointer = value_type*;
using const_pointer = const value_type*;
using iterator = NodeIterator;
using const_iterator = const NodeIterator;
using reverse_iterator = iterator;
using const_reverse_iterator = const_iterator;
using iterator_category = std::bidirectional_iterator_tag;
NodeIterator(uintptr_t Base, uintptr_t IRBase) : BaseList {Base}, IRList{ IRBase } {}
explicit NodeIterator(uintptr_t Base, uintptr_t IRBase, OrderedNodeWrapper Ptr) : BaseList {Base}, IRList{ IRBase }, Node {Ptr} {}
[[nodiscard]] bool operator==(const NodeIterator &rhs) const {
return Node.NodeOffset == rhs.Node.NodeOffset;
}
[[nodiscard]] bool operator!=(const NodeIterator &rhs) const {
return !operator==(rhs);
}
NodeIterator operator++() {
OrderedNodeHeader *RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
Node = RealNode->Next;
return *this;
}
NodeIterator operator--() {
OrderedNodeHeader *RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
Node = RealNode->Previous;
return *this;
}
[[nodiscard]] value_type operator*() {
OrderedNode *RealNode = Node.GetNode(BaseList);
return { RealNode, RealNode->Op(IRList) };
}
[[nodiscard]] value_type operator()() {
OrderedNode *RealNode = Node.GetNode(BaseList);
return { RealNode, RealNode->Op(IRList) };
}
[[nodiscard]] NodeID ID() const {
return Node.ID();
}
[[nodiscard]] static NodeIterator Invalid() {
return NodeIterator(0, 0);
}
protected:
uintptr_t BaseList{};
uintptr_t IRList{};
OrderedNodeWrapper Node{};
};
// This must directly match bytes to the named opsize.
// Implicit sized IR operations does math to get between sizes.
enum OpSize : uint8_t {
i8Bit = 1,
i16Bit = 2,
i32Bit = 4,
i64Bit = 8,
i128Bit = 16,
i256Bit = 32,
};
enum class FloatCompareOp : uint8_t {
EQ = 0,
LT,
LE,
UNO,
NEQ,
ORD,
};
enum class ShiftType : uint8_t {
LSL = 0,
LSR,
ASR,
ROR,
};
// Converts a size stored as an integer in to an OpSize enum.
// This is a nop operation and will be eliminated by the compiler.
static inline OpSize SizeToOpSize(uint8_t Size) {
switch (Size) {
case 1: return OpSize::i8Bit;
case 2: return OpSize::i16Bit;
case 4: return OpSize::i32Bit;
case 8: return OpSize::i64Bit;
case 16: return OpSize::i128Bit;
case 32: return OpSize::i256Bit;
default: FEX_UNREACHABLE;
}
}
#define IROP_ENUM
#define IROP_STRUCTS
#define IROP_SIZES
#define IROP_REG_CLASSES
#include <FEXCore/IR/IRDefines.inc>
/* This iterator can be used to step though every single node in a multi-block in SSA order.
*
* Iterates in the order of:
*
* end <-- CodeBlockA <--> BlockAInst1 <--> BlockAInst2 <--> CodeBlockB <--> BlockBInst1 <--> BlockBInst2 --> end
*/
class AllNodesIterator : public NodeIterator {
public:
AllNodesIterator(uintptr_t Base, uintptr_t IRBase) : NodeIterator(Base, IRBase) {}
explicit AllNodesIterator(uintptr_t Base, uintptr_t IRBase, OrderedNodeWrapper Ptr) : NodeIterator(Base, IRBase, Ptr) {}
AllNodesIterator(NodeIterator other) : NodeIterator(other) {} // Allow NodeIterator to be upgraded
AllNodesIterator operator++() {
OrderedNodeHeader *RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
auto IROp = Node.GetNode(BaseList)->Op(IRList);
// If this is the last node of a codeblock, we need to continue to the next block
if (IROp->Op == OP_ENDBLOCK) {
auto EndBlock = IROp->C<IROp_EndBlock>();
auto CurrentBlock = EndBlock->BlockHeader.GetNode(BaseList);
Node = CurrentBlock->Header.Next;
} else if (IROp->Op == OP_CODEBLOCK) {
auto CodeBlock = IROp->C<IROp_CodeBlock>();
Node = CodeBlock->Begin;
} else {
Node = RealNode->Next;
}
return *this;
}
AllNodesIterator operator--() {
auto IROp = Node.GetNode(BaseList)->Op(IRList);
if (IROp->Op == OP_BEGINBLOCK) {
auto BeginBlock = IROp->C<IROp_EndBlock>();
Node = BeginBlock->BlockHeader;
} else if (IROp->Op == OP_CODEBLOCK) {
auto PrevBlockWrapper = Node.GetNode(BaseList)->Header.Previous;
auto PrevCodeBlock = PrevBlockWrapper.GetNode(BaseList)->Op(IRList)->C<IROp_CodeBlock>();
Node = PrevCodeBlock->Last;
} else {
Node = Node.GetNode(BaseList)->Header.Previous;
}
return *this;
}
[[nodiscard]] static AllNodesIterator Invalid() {
return AllNodesIterator(0, 0);
}
};
class IRListView;
class IREmitter;
template<typename Type>
inline NodeID NodeWrapperBase<Type>::ID() const {
return NodeID(NodeOffset / sizeof(IR::OrderedNode));
}
bool IsFragmentExit(FEXCore::IR::IROps Op);
bool IsBlockExit(FEXCore::IR::IROps Op);
void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData);
fextl::unique_ptr<IREmitter> Parse(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator, fextl::stringstream &MapsStream);
}
template <>
struct std::hash<FEXCore::IR::NodeID> {
size_t operator()(const FEXCore::IR::NodeID& ID) const noexcept {
return std::hash<FEXCore::IR::NodeID::value_type>{}(ID.Value);
}
};
template <>
struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID::value_type> {
using Base = fmt::formatter<FEXCore::IR::NodeID::value_type>;
// Pass-through the underlying value, so IDs can
// be formatted like any integral value.
template <typename FormatContext>
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
return Base::format(ID.Value, ctx);
}
};
template <>
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
template <typename FormatContext>
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
return Base::format(Class.Val, ctx);
}
};
template <>
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
template <typename FormatContext>
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
return Base::format(Fence.Val, ctx);
}
};
template <>
struct fmt::formatter<FEXCore::IR::OpSize> : fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>> {
using Base = fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>>;
template <typename FormatContext>
auto format(const FEXCore::IR::OpSize& OpSize, FormatContext& ctx) const {
return Base::format(FEXCore::ToUnderlying(OpSize), ctx);
}
};
+10 -129
View File
@@ -347,7 +347,8 @@
"Desc": ["Loads a value from the static-ra context with offset",
"Dest = Ctx[Offset]"
],
"DestSize": "Size"
"DestSize": "Size",
"DynamicDispatch": true
},
"StoreRegister SSA:$Value, i1:$IsPrewrite, u32:$Offset, RegisterClass:$Class, RegisterClass:$StaticClass, u8:#Size": {
@@ -358,6 +359,7 @@
"Truncates if value's type is too large"
],
"DestSize": "Size",
"DynamicDispatch": true,
"EmitValidation": [
"WalkFindRegClass($Value) == $Class"
]
@@ -456,13 +458,6 @@
"DestSize": "4"
},
"GPR = LoadDF": {
"Desc": ["Loads the decimal flag from the context object in -1/1",
"representation for easy consumption"
],
"DestSize": "8"
},
"GPR = LoadFlag u32:$Flag": {
"Desc": ["Loads an x86-64 flag from the context object",
"Specialized to allow flexible implementation of flag handling"
@@ -603,15 +598,6 @@
"Ensures the memory operations are globally visible"
],
"HasSideEffects": true
},
"Prefetch i1:$ForStore, i1:$Stream, i8:$CacheLevel, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
"Desc": ["Does a cacheline prefetch operation"
],
"EmitValidation": [
"_CacheLevel > 0 && _CacheLevel < 4"
],
"HasSideEffects": true,
"DestSize": "8"
}
},
"Atomic": {
@@ -642,6 +628,7 @@
],
"HasDest": true,
"DestSize": "Size",
"ImplicitFlagClobber": true,
"NumElements": "2",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
@@ -966,53 +953,12 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Adc OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": [ "Integer Add with carry",
"Will truncate to 64 or 32bits"
],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Sbb OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": [ "Integer Subtract with carry/borrow",
"Will truncate to 64 or 32bits"
],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = AddShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
"Desc": [ "Integer Add with shifted register",
"Will truncate to 64 or 32bits"
],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
"_Shift != ShiftType::ROR"
]
},
"GPR = AddWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": [ "Integer add. Truncates and sets NZCV per AddNZCV"],
"DestSize": "Size",
"HasSideEffects": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"AddNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the sum of two GPRs"],
"HasSideEffects": true,
"DestSize": "Size"
},
"SetSmallNZV OpSize:#Size, GPR:$Src": {
"Desc": ["Set NZV with a SETF instruction. Preserves CF."],
"HasSideEffects": true,
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit"
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"CarryInvert": {
@@ -1035,30 +981,6 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"CondSubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2, CondClass:$Cond, u8:$FalseNZCV": {
"Desc": ["If condition is true, set NZCV per difference of GPRs, else force NZCV to a constant."],
"HasSideEffects": true,
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = AdcWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Adds and set NZCV for the sum of two GPRs and carry-in given as NZCV"],
"HasSideEffects": true,
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = SbbWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Subtracts and set NZCV for the difference of two GPRs and carry-in given as NZCV"],
"HasSideEffects": true,
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"AdcNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the sum of two GPRs and carry-in given as NZCV"],
"HasSideEffects": true,
@@ -1094,25 +1016,15 @@
"_Shift != ShiftType::ROR"
]
},
"GPR = SubWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": [ "Integer Sub. Truncates and sets NZCV per SubNZCV"],
"DestSize": "Size",
"HasSideEffects": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"CmpPairZ OpSize:#Size, GPRPair:$Src1, GPRPair:$Src2": {
"Desc": ["Compares register pairs and sets Z accordingly, preserving N/Z/V.",
"This accelerates cmpxchg."],
"HasSideEffects": true
},
"SubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the difference of two GPRs. ",
"Carry flag uses arm64 definition, inverted x86.",
""],
"DestSize": "Size",
"HasSideEffects": true
"HasSideEffects": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Or OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer binary or"
@@ -1161,13 +1073,6 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = XornShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
"Desc": [ "Integer binary exclusive or not with shifted register"],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = And OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer binary and"
],
@@ -1176,12 +1081,6 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = AndWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer binary and"
],
"DestSize": "Size",
"HasSideEffects": true
},
"GPR = Andn OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer binary AND NOT. Performs the equivalent of Src1 & ~Src2"],
"DestSize": "Size",
@@ -1242,18 +1141,6 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = UMull GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer unsigned multiplication long",
"Multiplies two 32-bit numbers, returning a 64-bit destination register."
],
"DestSize": "FEXCore::IR::OpSize::i64Bit"
},
"GPR = SMull GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer signed multiplication long",
"Multiplies two 32-bit numbers, returning a 64-bit destination register."
],
"DestSize": "FEXCore::IR::OpSize::i64Bit"
},
"GPR = Div OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer signed division"
],
@@ -1705,13 +1592,7 @@
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VUMaxV u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
"Desc": ["Does a horizontal vector unsigned maximum of elements across the source vector",
"Result is a zero extended scalar"
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFAbs u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
+2 -3
View File
@@ -6,10 +6,9 @@ tags: ir|dumper
$end_info$
*/
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/fextl/sstream.h>
#include <algorithm>
+2 -2
View File
@@ -6,9 +6,9 @@ tags: ir|emitter
$end_info$
*/
#include "Interface/IR/IREmitter.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
+4 -4
View File
@@ -6,9 +6,9 @@ tags: ir|parser
$end_info$
*/
#include "Interface/IR/IREmitter.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/StringUtils.h>
#include <FEXCore/fextl/sstream.h>
@@ -505,8 +505,8 @@ class IRParser: public FEXCore::IR::IREmitter {
}
if (Def.HasArgs) {
RemainingLine =
FEXCore::StringUtils::Trim(RemainingLine.substr(CurrentPos));
RemainingLine = FEXCore::StringUtils::Trim(RemainingLine.substr(CurrentPos));
CurrentPos = 0;
if (RemainingLine.empty()) {
// How did we get here?
Def.HasArgs = false;
+6 -6
View File
@@ -66,7 +66,7 @@ void PassManager::Finalize() {
}
}
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants) {
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants, bool StaticRegisterAllocation) {
FEX_CONFIG_OPT(DisablePasses, O0);
if (!DisablePasses()) {
@@ -80,10 +80,9 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool Inli
InsertPass(CreateDeadStoreElimination(ctx->HostFeatures.SupportsAVX));
InsertPass(CreatePassDeadCodeElimination());
InsertPass(CreateConstProp(
InlineConstants, ctx->HostFeatures.SupportsTSOImm9, Is64BitMode()));
InsertPass(CreateConstProp(InlineConstants, ctx->HostFeatures.SupportsTSOImm9));
InsertPass(CreateDeadFlagCalculationEliminination());
////// InsertPass(CreateDeadFlagCalculationEliminination());
InsertPass(CreateInlineCallOptimization(&ctx->CPUID));
InsertPass(CreatePassDeadCodeElimination());
@@ -102,8 +101,8 @@ void PassManager::AddDefaultValidationPasses() {
#endif
}
void PassManager::InsertRegisterAllocationPass(bool SupportsAVX) {
InsertPass(IR::CreateRegisterAllocationPass(GetPass("Compaction"), SupportsAVX), "RA");
void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAVX) {
InsertPass(IR::CreateRegisterAllocationPass(GetPass("Compaction"), OptimizeSRA, SupportsAVX), "RA");
}
bool PassManager::Run(IREmitter *IREmit) {
@@ -122,4 +121,5 @@ bool PassManager::Run(IREmitter *IREmit) {
return Changed;
}
}
+9 -2
View File
@@ -29,6 +29,8 @@ namespace FEXCore::IR {
class PassManager;
class IREmitter;
using ShouldExitHandler = std::function<void(void)>;
class Pass {
public:
virtual ~Pass() = default;
@@ -45,7 +47,7 @@ protected:
class PassManager final {
friend class InlineCallOptimization;
public:
void AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants);
void AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants, bool StaticRegisterAllocation);
void AddDefaultValidationPasses();
Pass* InsertPass(fextl::unique_ptr<Pass> Pass, fextl::string Name = "") {
auto PassPtr = InsertAt(Passes.end(), std::move(Pass))->get();
@@ -56,10 +58,14 @@ public:
return PassPtr;
}
void InsertRegisterAllocationPass(bool SupportsAVX);
void InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAVX);
bool Run(IREmitter *IREmit);
void RegisterExitHandler(ShouldExitHandler Handler) {
ExitHandler = std::move(Handler);
}
bool HasPass(fextl::string Name) const {
return NameToPassMaping.contains(Name);
}
@@ -80,6 +86,7 @@ public:
void Finalize();
protected:
ShouldExitHandler ExitHandler;
FEXCore::HLE::SyscallHandler *SyscallHandler;
private:
+4 -5
View File
@@ -16,17 +16,16 @@ class Pass;
class RegisterAllocationPass;
class RegisterAllocationData;
fextl::unique_ptr<FEXCore::IR::Pass>
CreateConstProp(bool InlineConstants, bool SupportsTSOImm9, bool Is64BitMode);
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants, bool SupportsTSOImm9);
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX);
fextl::unique_ptr<FEXCore::IR::Pass> CreateInlineCallOptimization(const FEXCore::CPUIDEmu* CPUID);
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination(bool SupportsAVX);
fextl::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination();
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass>
CreateRegisterAllocationPass(FEXCore::IR::Pass *CompactionPass,
bool SupportsAVX);
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass,
bool OptimizeSRA,
bool SupportsAVX);
fextl::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
namespace Validation {
+123 -254
View File
@@ -13,10 +13,11 @@ $end_info$
#include "aarch64/disasm-aarch64.h"
#include "aarch64/assembler-aarch64.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/map.h>
@@ -26,7 +27,6 @@ $end_info$
#include <bit>
#include <cstdint>
#include <memory>
#include <optional>
#include <string.h>
#include <tuple>
#include <utility>
@@ -90,132 +90,58 @@ static bool IsTSOImm9(uint64_t imm) {
}
}
using MemExtendedAddrResult =
std::tuple<MemOffsetType, uint8_t, OrderedNode *, OrderedNode *>;
// If this optimization doesn't succeed, it will return the nullopt
static std::optional<MemExtendedAddrResult>
MemExtendedAddressing(IREmitter *IREmit, uint8_t AccessSize,
IROp_Header *AddressHeader) {
// Try to optimize: AddShift Base, LSHL(Offset, Scale)
if (AddressHeader->Op == OP_ADDSHIFT) {
auto AddShift = AddressHeader->C<IROp_AddShift>();
if (AddShift->Shift == IR::ShiftType::LSL) {
auto Scale = 1U << AddShift->ShiftAmount;
if (IsMemoryScale(Scale, AccessSize)) {
// remove shift as it can be folded to the mem op
return std::make_optional(
std::make_tuple(MEM_OFFSET_SXTX, (uint8_t)Scale,
IREmit->UnwrapNode(AddShift->Src2),
IREmit->UnwrapNode(AddShift->Src1)));
} else if (Scale == 1) {
return std::make_optional(std::make_tuple(
MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddShift->Src2),
IREmit->UnwrapNode(AddShift->Src1)));
}
}
return std::nullopt;
}
LOGMAN_THROW_A_FMT(AddressHeader->Op == OP_ADD, "Invalid address Op");
static std::tuple<MemOffsetType, uint8_t, OrderedNode*, OrderedNode*> MemExtendedAddressing(IREmitter *IREmit, uint8_t AccessSize, IROp_Header* AddressHeader) {
auto Src0Header = IREmit->GetOpHeader(AddressHeader->Args[0]);
if (Src0Header->Size == 8) {
// Try to optimize: Base + MUL(Offset, Scale)
//Try to optimize: Base + MUL(Offset, Scale)
if (Src0Header->Op == OP_MUL) {
uint64_t Scale;
if (IREmit->IsValueConstant(Src0Header->Args[1], &Scale)) {
if (IsMemoryScale(Scale, AccessSize)) {
// remove mul as it can be folded to the mem op
return std::make_optional(
std::make_tuple(MEM_OFFSET_SXTX, (uint8_t)Scale,
IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
return { MEM_OFFSET_SXTX, (uint8_t)Scale, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
} else if (Scale == 1) {
// remove nop mul
return std::make_optional(std::make_tuple(
MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
}
}
}
// Try to optimize: Base + LSHL(Offset, Scale)
//Try to optimize: Base + LSHL(Offset, Scale)
else if (Src0Header->Op == OP_LSHL) {
uint64_t Constant2;
if (IREmit->IsValueConstant(Src0Header->Args[1], &Constant2)) {
uint64_t Scale = 1<<Constant2;
if (IsMemoryScale(Scale, AccessSize)) {
// remove shift as it can be folded to the mem op
return std::make_optional(
std::make_tuple(MEM_OFFSET_SXTX, Scale,
IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
return { MEM_OFFSET_SXTX, Scale, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
} else if (Scale == 1) {
// remove nop shift
return std::make_optional(std::make_tuple(
MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
}
}
}
#if defined(_M_ARM_64) // x86 can't sext or zext on mem ops
// Try to optimize: Base + (u32)Offset
//Try to optimize: Base + (u32)Offset
else if (Src0Header->Op == OP_BFE) {
auto Bfe = Src0Header->C<IROp_Bfe>();
if (Bfe->lsb == 0 && Bfe->Width == 32) {
//todo: arm can also scale here
return std::make_optional(std::make_tuple(
MEM_OFFSET_UXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
return { MEM_OFFSET_UXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
}
}
// Try to optimize: Base + (s32)Offset
//Try to optimize: Base + (s32)Offset
else if (Src0Header->Op == OP_SBFE) {
auto Sbfe = Src0Header->C<IROp_Sbfe>();
if (Sbfe->lsb == 0 && Sbfe->Width == 32) {
// todo: arm can also scale here
return std::make_optional(std::make_tuple(
MEM_OFFSET_SXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]),
IREmit->UnwrapNode(Src0Header->Args[0])));
//todo: arm can also scale here
return { MEM_OFFSET_SXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) };
}
}
#endif
}
// no match anywhere, just add
// However, if we have one 32bit negative constant, we need to sign extend it
auto Arg0_ = AddressHeader->Args[0];
auto Arg1_ = AddressHeader->Args[1];
auto Arg1H = IREmit->GetOpHeader(Arg1_);
auto Arg0 = IREmit->UnwrapNode(Arg0_);
auto Arg1 = IREmit->UnwrapNode(Arg1_);
uint64_t ConstVal = 0;
// Only optimize in 32bits reg+const where const < 16Kb.
if (Arg1H->Size == 4 && IREmit->IsValueConstant(Arg1_, &ConstVal)) {
// Base is Arg0, Constant (Displacement in Arg1)
OrderedNode *Base = Arg0;
OrderedNode *Cnt = Arg1;
int32_t Val32 = (int32_t)ConstVal;
if (Val32 > -16384 && Val32 < 0) {
return std::make_optional(std::make_tuple(MEM_OFFSET_SXTW, 1, Base, Cnt));
} else if (Val32 >= 0 && Val32 < 16384) {
return std::make_optional(std::make_tuple(MEM_OFFSET_SXTX, 1, Base, Cnt));
}
} else if (AddressHeader->Size == 4) {
// Do not optimize 32bit reg+reg.
// Something like :
// add w20, w7, w5
// ldr w7, [x20]
//
// cannot be simplified to (or any other single load instruction)
// ldr w7, [x5, w7, sxtx]
return std::nullopt;
} else {
return std::make_optional(std::make_tuple(MEM_OFFSET_SXTX, 1, Arg0, Arg1));
}
return std::nullopt;
return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[0]), IREmit->UnwrapNode(AddressHeader->Args[1]) };
}
static OrderedNodeWrapper RemoveUselessMasking(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t mask) {
@@ -258,10 +184,9 @@ static bool IsBfeAlreadyDone(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t
class ConstProp final : public FEXCore::IR::Pass {
public:
explicit ConstProp(bool DoInlineConstants, bool SupportsTSOImm9,
bool Is64BitMode)
: InlineConstants(DoInlineConstants), SupportsTSOImm9{SupportsTSOImm9},
Is64BitMode(Is64BitMode) {}
explicit ConstProp(bool DoInlineConstants, bool SupportsTSOImm9)
: InlineConstants(DoInlineConstants)
, SupportsTSOImm9 {SupportsTSOImm9} { }
bool Run(IREmitter *IREmit) override;
@@ -294,7 +219,6 @@ private:
return Result.first->second;
}
bool SupportsTSOImm9{};
bool Is64BitMode;
// This is a heuristic to limit constant pool live ranges to reduce RA interference pressure.
// If the range is unbounded then RA interference pressure seems to increase to the point
// that long blocks of constant usage can slow to a crawl.
@@ -515,6 +439,50 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
bool Changed = false;
switch (IROp->Op) {
/*
case OP_UMUL:
case OP_DIV:
case OP_UDIV:
case OP_REM:
case OP_UREM:
case OP_MULH:
case OP_UMULH:
case OP_LSHR:
case OP_ASHR:
case OP_ROL:
case OP_ROR:
case OP_LDIV:
case OP_LUDIV:
case OP_LREM:
case OP_LUREM:
case OP_BFI:
{
uint64_t Constant1;
uint64_t Constant2;
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) &&
IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
LOGMAN_MSG_A_FMT("Could const prop op: {}", IR::GetName(IROp->Op));
}
break;
}
case OP_SEXT:
case OP_NEG:
case OP_POPCOUNT:
case OP_FINDLSB:
case OP_FINDMSB:
case OP_REV:
case OP_SBFE: {
uint64_t Constant1;
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
LOGMAN_MSG_A_FMT("Could const prop op: {}", IR::GetName(IROp->Op));
}
break;
}
*/
case OP_LOADMEMTSO: {
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
@@ -522,12 +490,8 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
// Support once hardware is available to use this.
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
@@ -545,12 +509,8 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
// Support once hardware is available to use this.
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
@@ -565,39 +525,8 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
auto Op = IROp->CW<IR::IROp_LoadMem>();
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
if (AddressHeader->Op == OP_ADD &&
((Is64BitMode && AddressHeader->Size == 8) ||
(!Is64BitMode && AddressHeader->Size == 4))) {
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
Changed = true;
}
break;
}
case OP_STOREMEM: {
auto Op = IROp->CW<IR::IROp_StoreMem>();
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
if (AddressHeader->Op == OP_ADD &&
((Is64BitMode && AddressHeader->Size == 8) ||
(!Is64BitMode && AddressHeader->Size == 4))) {
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
@@ -609,27 +538,16 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
break;
}
case OP_PREFETCH: {
auto Op = IROp->CW<IR::IROp_Prefetch>();
case OP_STOREMEM: {
auto Op = IROp->CW<IR::IROp_StoreMem>();
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
const bool SupportedOp =
AddressHeader->Op == OP_ADD ||
AddressHeader->Op == OP_ADDSHIFT;
if (SupportedOp &&
((Is64BitMode && AddressHeader->Size == 8) ||
(!Is64BitMode && AddressHeader->Size == 4))) {
auto MaybeMemAddr =
MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
if (!MaybeMemAddr) {
break;
}
auto [OffsetType, OffsetScale, Arg0, Arg1] = *MaybeMemAddr;
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
Changed = true;
@@ -637,37 +555,23 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
break;
}
case OP_ADD:
case OP_SUB:
case OP_ADDWITHFLAGS:
case OP_SUBWITHFLAGS: {
case OP_ADD: {
auto Op = IROp->C<IR::IROp_Add>();
uint64_t Constant1{};
uint64_t Constant2{};
bool IsConstant1 = IREmit->IsValueConstant(Op->Header.Args[0], &Constant1);
bool IsConstant2 = IREmit->IsValueConstant(Op->Header.Args[1], &Constant2);
if (IsConstant1 && IsConstant2 && IROp->Op == OP_ADD) {
if (IsConstant1 && IsConstant2) {
uint64_t NewConstant = (Constant1 + Constant2) & getMask(Op) ;
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
Changed = true;
} else if (IsConstant1 && IsConstant2 && IROp->Op == OP_SUB) {
uint64_t NewConstant = (Constant1 - Constant2) & getMask(Op) ;
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
Changed = true;
}
else if (IsConstant2 && !IsImmAddSub(Constant2) && IsImmAddSub(-Constant2)) {
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
// So, negate the operation to negate (and inline) the constant.
if (IROp->Op == OP_ADD)
IROp->Op = OP_SUB;
else if (IROp->Op == OP_SUB)
IROp->Op = OP_ADD;
else if (IROp->Op == OP_ADDWITHFLAGS)
IROp->Op = OP_SUBWITHFLAGS;
else if (IROp->Op == OP_SUBWITHFLAGS)
IROp->Op = OP_ADDWITHFLAGS;
// This means we can convert the operation in to a subtract.
// Change the IR operation itself.
IROp->Op = OP_SUB;
// Set the write cursor to just before this operation.
auto CodeIter = CurrentIR.at(CodeNode);
--CodeIter;
@@ -682,6 +586,19 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
}
break;
}
case OP_SUB: {
auto Op = IROp->C<IR::IROp_Sub>();
uint64_t Constant1{};
uint64_t Constant2{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
uint64_t NewConstant = (Constant1 - Constant2) & getMask(Op) ;
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
Changed = true;
}
break;
}
case OP_SUBSHIFT: {
auto Op = IROp->C<IR::IROp_SubShift>();
@@ -728,6 +645,23 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
}
break;
}
/* TODO: restore this when we have rmif or something? */
#if 0
case OP_TESTNZ: {
auto Op = IROp->CW<IR::IROp_TestNZ>();
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
bool N = Constant1 & (1ull << ((Op->Size * 8) - 1));
bool Z = Constant1 == 0;
uint32_t NZVC = (N ? (1u << 31) : 0) | (Z ? (1u << 30) : 0);
IREmit->ReplaceWithConstant(CodeNode, NZVC);
Changed = true;
}
break;
}
#endif
case OP_OR: {
auto Op = IROp->CW<IR::IROp_Or>();
uint64_t Constant1{};
@@ -804,17 +738,6 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
}
break;
}
case OP_NEG: {
auto Op = IROp->CW<IR::IROp_Neg>();
uint64_t Constant{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant)) {
uint64_t NewConstant = -Constant;
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
Changed = true;
}
break;
}
case OP_LSHL: {
auto Op = IROp->CW<IR::IROp_Lshl>();
uint64_t Constant1{};
@@ -910,17 +833,12 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
uint64_t Constant;
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
// SBFE of a constant can be converted to a constant.
uint64_t SourceMask =
Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
uint64_t DestSizeInBits = IROp->Size * 8;
uint64_t DestMask =
DestSizeInBits == 64 ? ~0ULL : ((1ULL << DestSizeInBits) - 1);
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
SourceMask <<= Op->lsb;
int64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
NewConstant <<= 64 - Op->Width;
NewConstant >>= 64 - Op->Width;
NewConstant &= DestMask;
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
Changed = true;
@@ -1042,24 +960,20 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
case OP_SUB:
case OP_ADDNZCV:
case OP_SUBNZCV:
case OP_ADDWITHFLAGS:
case OP_SUBWITHFLAGS:
{
auto Op = IROp->C<IR::IROp_Add>();
uint64_t Constant2{};
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
// We don't allow 8/16-bit operations to have constants, since no
// constant would be in bounds after the JIT's 24/16 shift.
if (IsImmAddSub(Constant2) && Op->Header.Size >= 4) {
if (IsImmAddSub(Constant2)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
Changed = true;
}
} else if (IROp->Op == OP_SUBNZCV || IROp->Op == OP_SUBWITHFLAGS || IROp->Op == OP_SUB) {
// TODO: Generalize this
} else if (IROp->Op == OP_SUBNZCV) {
// If the first source is zero, we can use a NEGS instruction.
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
if (Constant1 == 0) {
@@ -1072,39 +986,7 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
break;
}
case OP_ADC:
case OP_ADCWITHFLAGS:
{
auto Op = IROp->C<IR::IROp_Adc>();
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
if (Constant1 == 0) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
Changed = true;
}
}
break;
}
case OP_RMIFNZCV:
{
auto Op = IROp->C<IR::IROp_RmifNZCV>();
uint64_t Constant1{};
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
if (Constant1 == 0) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
Changed = true;
}
}
break;
}
case OP_CONDADDNZCV:
case OP_CONDSUBNZCV:
{
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
@@ -1161,12 +1043,17 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
}
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
#ifdef JIT_ARM64
bool SupportsAllOnes = true;
#else
bool SupportsAllOnes = false;
#endif
uint64_t Constant2{};
uint64_t Constant3{};
if (IREmit->IsValueConstant(Op->Header.Args[2], &Constant2) &&
IREmit->IsValueConstant(Op->Header.Args[3], &Constant3) &&
(Constant2 == 1 || Constant2 == AllOnes) &&
(Constant2 == 1 || (SupportsAllOnes && Constant2 == AllOnes)) &&
Constant3 == 0)
{
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[2]));
@@ -1247,7 +1134,6 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
case OP_OR:
case OP_XOR:
case OP_AND:
case OP_ANDWITHFLAGS:
case OP_ANDN:
{
auto Op = IROp->CW<IR::IROp_Or>();
@@ -1340,7 +1226,7 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
Changed = true;
}
@@ -1354,29 +1240,13 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant));
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
Changed = true;
}
break;
}
case OP_PREFETCH:
{
auto Op = IROp->CW<IR::IROp_Prefetch>();
uint64_t Constant2{};
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) {
if (IsImmMemory(Constant2, IROp->Size)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset));
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, CreateInlineConstant(IREmit, Constant2));
Changed = true;
}
}
break;
}
default:
break;
}
@@ -1415,9 +1285,8 @@ bool ConstProp::Run(IREmitter *IREmit) {
return Changed;
}
fextl::unique_ptr<FEXCore::IR::Pass>
CreateConstProp(bool InlineConstants, bool SupportsTSOImm9, bool Is64BitMode) {
return fextl::make_unique<ConstProp>(InlineConstants, SupportsTSOImm9,
Is64BitMode);
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants, bool SupportsTSOImm9) {
return fextl::make_unique<ConstProp>(InlineConstants, SupportsTSOImm9);
}
}
@@ -5,10 +5,11 @@ tags: ir|opts
$end_info$
*/
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
@@ -6,14 +6,13 @@ desc: Transforms ContextLoad/Store to temporaries, similar to mem2reg
$end_info$
*/
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/EnumOperators.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
@@ -484,8 +483,6 @@ private:
ContextMemberInfo *RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode = nullptr);
ContextMemberInfo *RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode = nullptr);
bool HandleLoadFlag(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::OrderedNode *CodeNode, unsigned Flag);
// Classify context loads and stores.
bool ClassifyContextLoad(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::NodeIterator BlockEnd);
bool ClassifyContextStore(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::OrderedNode *ValueNode);
@@ -547,42 +544,9 @@ bool RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter *IREmit, ContextInfo *Loc
bool RCLSE::ClassifyContextStore(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::OrderedNode *ValueNode) {
auto Info = FindMemberInfo(LocalInfo, Offset, Size);
ContextMemberInfo PreviousMemberInfoCopy = *Info;
RecordAccess(Info, Class, Offset, Size, LastAccessType::WRITE, ValueNode,
CodeNode);
if (PreviousMemberInfoCopy.AccessRegClass == Info->AccessRegClass &&
PreviousMemberInfoCopy.AccessOffset == Info->AccessOffset &&
PreviousMemberInfoCopy.AccessSize == Size &&
PreviousMemberInfoCopy.Accessed == LastAccessType::WRITE) {
// This optimizes redundant stores with no intervening load
IREmit->Remove(PreviousMemberInfoCopy.StoreNode);
return true;
}
// TODO: Optimize the case of partial stores.
return false;
}
bool RCLSE::HandleLoadFlag(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::OrderedNode *CodeNode, unsigned Flag) {
const auto FlagOffset = offsetof(FEXCore::Core::CPUState, flags[Flag]);
auto Info = FindMemberInfo(LocalInfo, FlagOffset, 1);
LastAccessType LastAccess = Info->Accessed;
auto LastValueNode = Info->ValueNode;
if (IsWriteAccess(LastAccess)) { // 1 byte so always a full write
// If the last store matches this load value then we can replace the loaded value with the previous valid one
IREmit->SetWriteCursor(CodeNode);
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
return true;
}
else if (IsReadAccess(LastAccess)) {
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
return true;
}
Info = RecordAccess(Info, Class, Offset, Size, LastAccessType::WRITE, ValueNode, CodeNode);
// TODO: Optimize redundant stores.
// ContextMemberInfo PreviousMemberInfoCopy = *Info;
return false;
}
@@ -695,11 +659,23 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
}
else if (IROp->Op == OP_LOADFLAG) {
const auto Op = IROp->CW<IR::IROp_LoadFlag>();
const auto FlagOffset = offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag;
auto Info = FindMemberInfo(&LocalInfo, FlagOffset, 1);
LastAccessType LastAccess = Info->Accessed;
OrderedNode *LastValueNode = Info->ValueNode;
Changed |= HandleLoadFlag(IREmit, &LocalInfo, CodeNode, Op->Flag);
}
else if (IROp->Op == OP_LOADDF) {
Changed |= HandleLoadFlag(IREmit, &LocalInfo, CodeNode, X86State::RFLAG_DF_RAW_LOC);
if (IsWriteAccess(LastAccess)) { // 1 byte so always a full write
// If the last store matches this load value then we can replace the loaded value with the previous valid one
IREmit->SetWriteCursor(CodeNode);
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
Changed = true;
}
else if (IsReadAccess(LastAccess)) {
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
Changed = true;
}
}
else if (IROp->Op == OP_SYSCALL ||
IROp->Op == OP_INLINESYSCALL) {
@@ -6,12 +6,12 @@ desc: Cross block store-after-store elimination
$end_info$
*/
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/unordered_map.h>
@@ -211,10 +211,6 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
auto& BlockInfo = InfoMap[BlockNode];
BlockInfo.flag.reads |= 1UL << Op->Flag;
} else if (IROp->Op == OP_LOADDF) {
auto& BlockInfo = InfoMap[BlockNode];
BlockInfo.flag.reads |= 1UL << X86State::RFLAG_DF_RAW_LOC;
} else if (IROp->Op == OP_STOREREGISTER) {
auto Op = IROp->C<IR::IROp_StoreRegister>();
@@ -6,11 +6,12 @@ desc: Sorts the ssa storage in memory, needed for RA and others
$end_info$
*/
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/Profiler.h>
@@ -6,8 +6,6 @@ desc: Prints IR
$end_info$
*/
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Interface/Core/OpcodeDispatcher.h"
@@ -6,14 +6,14 @@ desc: Sanity checking pass
$end_info$
*/
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Interface/IR/Passes/IRValidation.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/sstream.h>
@@ -7,10 +7,11 @@ $end_info$
*/
#include "Interface/Core/CPUID.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Profiler.h>
@@ -6,9 +6,10 @@ desc: Long divide elimination pass
$end_info$
*/
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
@@ -1,13 +1,12 @@
// SPDX-License-Identifier: MIT
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Interface/IR/Passes/IRValidation.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/deque.h>
#include <FEXCore/fextl/fmt.h>
@@ -6,10 +6,9 @@ desc: This is not used right now, possibly broken
$end_info$
*/
#include "FEXCore/Core/X86Enums.h"
#include "Interface/IR/IREmitter.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/Profiler.h>
#include "Interface/IR/PassManager.h"
@@ -17,381 +16,54 @@ $end_info$
#include <array>
#include <memory>
// Flag bit flags
#define FLAG_V (1U << 0)
#define FLAG_C (1U << 1)
#define FLAG_Z (1U << 2)
#define FLAG_N (1U << 3)
#define FLAG_A (1U << 4)
#define FLAG_P (1U << 5)
#define FLAG_ZCV (FLAG_Z | FLAG_C | FLAG_V)
#define FLAG_NZCV (FLAG_N | FLAG_ZCV)
#define FLAG_ALL (FLAG_NZCV | FLAG_A | FLAG_P)
namespace FEXCore::IR {
struct FlagInfo {
// If set, all following fields are zero, used for a quick exit.
bool Trivial;
// Set of flags read by the instruction.
unsigned Read;
// Set of flags written by the instruction. Happens AFTER the reads.
unsigned Write;
// If true, the instruction can be be eliminated if its flag writes can all be
// eliminated.
bool CanEliminate;
// If true, the opcode can be replaced with Replacement if its flag writes can
// all be eliminated.
bool CanReplace;
IROps Replacement;
};
class DeadFlagCalculationEliminination final : public FEXCore::IR::Pass {
public:
bool Run(IREmitter *IREmit) override;
private:
FlagInfo Classify(IROp_Header *Node);
unsigned FlagForOffset(unsigned Offset);
unsigned FlagsForCondClassType(CondClassType Cond);
};
unsigned
DeadFlagCalculationEliminination::FlagForOffset(unsigned Offset)
{
return Offset == offsetof(FEXCore::Core::CPUState, pf_raw) ? FLAG_P :
Offset == offsetof(FEXCore::Core::CPUState, af_raw) ? FLAG_A :
0;
};
unsigned
DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond)
{
switch (Cond) {
case COND_AL:
return 0;
case COND_MI:
case COND_PL:
return FLAG_N;
case COND_EQ:
case COND_NEQ:
return FLAG_Z;
case COND_UGE:
case COND_ULT:
return FLAG_C;
case COND_VS:
case COND_VC:
case COND_FU:
case COND_FNU:
return FLAG_V;
case COND_UGT:
case COND_ULE:
return FLAG_Z | FLAG_C;
case COND_SGE:
case COND_SLT:
case COND_FLU:
case COND_FGE:
return FLAG_N | FLAG_V;
case COND_SGT:
case COND_SLE:
case COND_FLEU:
case COND_FGT:
return FLAG_N | FLAG_Z | FLAG_V;
default:
LOGMAN_THROW_AA_FMT(false, "unknown cond class type");
return FLAG_NZCV;
}
}
FlagInfo
DeadFlagCalculationEliminination::Classify(IROp_Header *IROp)
{
switch (IROp->Op) {
case OP_ANDWITHFLAGS:
return {
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_AND,
};
case OP_ADDWITHFLAGS:
return {
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_ADD,
};
case OP_SUBWITHFLAGS:
return {
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_SUB,
};
case OP_ADCWITHFLAGS:
return {
.Read = FLAG_C,
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_ADC,
};
case OP_SBBWITHFLAGS:
return {
.Read = FLAG_C,
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_SBB,
};
case OP_ADDNZCV:
case OP_SUBNZCV:
case OP_TESTNZ:
case OP_FCMP:
case OP_STORENZCV:
return {
.Write = FLAG_NZCV,
.CanEliminate = true,
};
case OP_AXFLAG:
// Per the Arm spec, axflag reads Z/V/C but not N. It writes all flags.
return {
.Read = FLAG_ZCV,
.Write = FLAG_NZCV,
.CanEliminate = true,
};
case OP_CMPPAIRZ:
return {
.Write = FLAG_Z,
.CanEliminate = true,
};
case OP_CARRYINVERT:
return {
.Read = FLAG_C,
.Write = FLAG_C,
.CanEliminate = true,
};
case OP_SETSMALLNZV:
return {
.Write = FLAG_N | FLAG_Z | FLAG_V,
.CanEliminate = true,
};
case OP_LOADNZCV:
return {.Read = FLAG_NZCV};
case OP_ADC:
case OP_SBB:
return {.Read = FLAG_C};
case OP_ADCNZCV:
case OP_SBBNZCV:
return {
.Read = FLAG_C,
.Write = FLAG_NZCV,
.CanEliminate = true,
};
case OP_NZCVSELECT: {
auto Op = IROp->CW<IR::IROp_NZCVSelect>();
return {.Read = FlagsForCondClassType(Op->Cond)};
}
case OP_NEG: {
auto Op = IROp->CW<IR::IROp_Neg>();
return {.Read = FlagsForCondClassType(Op->Cond)};
}
case OP_CONDJUMP: {
auto Op = IROp->CW<IR::IROp_CondJump>();
if (!Op->FromNZCV)
break;
return {.Read = FlagsForCondClassType(Op->Cond)};
}
case OP_CONDSUBNZCV:
case OP_CONDADDNZCV: {
auto Op = IROp->CW<IR::IROp_CondAddNZCV>();
return {
.Read = FlagsForCondClassType(Op->Cond),
.Write = FLAG_NZCV,
.CanEliminate = true,
};
}
case OP_RMIFNZCV: {
auto Op = IROp->CW<IR::IROp_RmifNZCV>();
static_assert(FLAG_N == (1 << 3), "rmif mask lines up with our bits");
static_assert(FLAG_Z == (1 << 2), "rmif mask lines up with our bits");
static_assert(FLAG_C == (1 << 1), "rmif mask lines up with our bits");
static_assert(FLAG_V == (1 << 0), "rmif mask lines up with our bits");
return {
.Write = Op->Mask,
.CanEliminate = true,
};
}
case OP_INVALIDATEFLAGS: {
auto Op = IROp->CW<IR::IROp_InvalidateFlags>();
unsigned Flags = 0;
// TODO: Make this translation less silly
if (Op->Flags & (1u << X86State::RFLAG_SF_RAW_LOC))
Flags |= FLAG_N;
if (Op->Flags & (1u << X86State::RFLAG_ZF_RAW_LOC))
Flags |= FLAG_Z;
if (Op->Flags & (1u << X86State::RFLAG_CF_RAW_LOC))
Flags |= FLAG_C;
if (Op->Flags & (1u << X86State::RFLAG_OF_RAW_LOC))
Flags |= FLAG_V;
if (Op->Flags & (1u << X86State::RFLAG_PF_RAW_LOC))
Flags |= FLAG_P;
if (Op->Flags & (1u << X86State::RFLAG_AF_RAW_LOC))
Flags |= FLAG_A;
// The mental model of InvalidateFlags is writing undefined values to all
// of the selected flags, allowing the write-after-write optimizations to
// optimize invalidate-after-write for free.
return {
.Write = Flags,
.CanEliminate = true,
};
}
case OP_LOADREGISTER: {
auto Op = IROp->CW<IR::IROp_LoadRegister>();
if (Op->Class != GPRClass || Op->StaticClass != GPRFixedClass)
break;
return {.Read = FlagForOffset(Op->Offset)};
}
case OP_STOREREGISTER: {
auto Op = IROp->CW<IR::IROp_StoreRegister>();
if (Op->Class != GPRClass || Op->StaticClass != GPRFixedClass)
break;
LOGMAN_THROW_A_FMT(!Op->IsPrewrite, "PF/AF writes are fixed-form");
unsigned Flag = FlagForOffset(Op->Offset);
return {
.Write = Flag,
.CanEliminate = Flag != 0,
};
}
default:
break;
}
return {.Trivial = true};
}
/**
* @brief This pass removes flag calculations that will otherwise be unused INSIDE of that block
* @brief (UNSAFE) This pass removes flag calculations that will otherwise be unused INSIDE of that block
*
* Compilers don't really do any form of cross-block flag allocation like they do RA with GPRs.
* This ends up with them recalculating flags across blocks regardless of if it is actually possible to reuse the flags.
* This is an additional burden in x86 that most instructions change flags when called, so it is easier to recalculate anyway.
*
* This is unsafe since handwritten code can easily break this assumption.
* This may be more interesting with full function level recompilation since flags definitely won't be used across function boundaries.
*
*/
bool DeadFlagCalculationEliminination::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
std::array<OrderedNode*, 32> LastValidFlagStores{};
bool Changed = false;
auto CurrentIR = IREmit->ViewIR();
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
// We model all flags as read at the end of the block, since this pass is
// presently purely local. Optimizing this requires global anslysis.
uint32_t FlagsRead = FLAG_ALL;
// Reverse iteration is not yet working with the iterators
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
// We grab these nodes this way so we can iterate easily
auto CodeBegin = CurrentIR.at(BlockIROp->Begin);
auto CodeLast = CurrentIR.at(BlockIROp->Last);
// Iterate the block in reverse
while (1) {
auto [CodeNode, IROp] = CodeLast();
// Optimizing flags can cause earlier flag reads to become dead but dead
// flag reads should not impede optimiation of earlier dead flag writes.
// We must DCE as we go to ensure we converge in a single iteration.
//
// TODO: This whole pass could be merged with DCE?
bool HasSideEffects = IR::HasSideEffects(IROp->Op);
if (!HasSideEffects && CodeNode->GetUses() == 0) {
Changed = true;
IREmit->Remove(CodeNode);
} else {
// Optimiation algorithm: For each flag written...
//
// If the flag has a later read (per FlagsRead), remove the flag from
// FlagsRead, since the reader is covered by this write.
//
// Else, there is no later read, so remove the flag write (if we can).
// This is the active part of the optimization.
//
// Then, add each flag read to FlagsRead.
//
// This order is important: instructions that read-modify-write flags
// (like adcs) first read flags, then write flags. Since we're iterating
// the block backwards, that means we handle the write first.
struct FlagInfo Info = Classify(IROp);
if (!Info.Trivial) {
bool Eliminated = false;
if ((FlagsRead & Info.Write) == 0) {
if (Info.CanEliminate) {
IREmit->Remove(CodeNode);
Eliminated = true;
Changed = true;
} else if (Info.CanReplace) {
IROp->Op = Info.Replacement;
Changed = true;
}
} else {
FlagsRead &= ~Info.Write;
}
// If we eliminated the instruction, we eliminate its read too. This
// check is required to ensure the pass converges locally in a single
// iteration.
if (!Eliminated)
FlagsRead |= Info.Read;
}
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
if (IROp->Op == OP_STOREFLAG) {
auto Op = IROp->CW<IR::IROp_StoreFlag>();
// Set this node as the last one valid for this flag
LastValidFlagStores[Op->Flag] = CodeNode;
}
// Iterate in reverse
if (CodeLast == CodeBegin) {
break;
else if (IROp->Op == OP_LOADFLAG) {
auto Op = IROp->CW<IR::IROp_LoadFlag>();
LastValidFlagStores[Op->Flag] = nullptr;
}
--CodeLast;
}
// If any flags are stored but not loaded by the end of the block, then erase them
for (auto &Flag : LastValidFlagStores) {
if (Flag != nullptr) {
IREmit->Remove(Flag);
Changed = true;
}
}
LastValidFlagStores.fill(nullptr);
}
return Changed;
@@ -5,27 +5,26 @@ tags: ir|opts
$end_info$
*/
#include "Utils/BucketList.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "FEXCore/Core/X86Enums.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Interface/IR/Passes.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/BucketList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/unordered_set.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <algorithm>
#include <cstddef>
@@ -68,7 +67,7 @@ namespace {
};
static_assert(sizeof(RegisterNode) == 128 * 4);
constexpr size_t REGISTER_NODES_PER_PAGE = FEXCore::Utils::FEX_PAGE_SIZE / sizeof(RegisterNode);
constexpr size_t REGISTER_NODES_PER_PAGE = FHU::FEX_PAGE_SIZE / sizeof(RegisterNode);
struct RegisterSet {
fextl::vector<RegisterClass> Classes;
@@ -224,7 +223,7 @@ namespace {
class ConstrainedRAPass final : public RegisterAllocationPass {
public:
ConstrainedRAPass(FEXCore::IR::Pass* _CompactionPass, bool SupportsAVX);
ConstrainedRAPass(FEXCore::IR::Pass* _CompactionPass, bool OptimizeSRA, bool SupportsAVX);
~ConstrainedRAPass();
bool Run(IREmitter *IREmit) override;
@@ -240,6 +239,8 @@ namespace {
RegisterAllocationData::UniquePtr PullAllocationData() override;
private:
using BlockInterferences = fextl::vector<IR::NodeID>;
IR::NodeID SpillPointId;
fextl::vector<BucketList<DEFAULT_INTERFERENCE_SPAN_COUNT, uint32_t>> SpanStart;
@@ -247,10 +248,14 @@ namespace {
RegisterGraph *Graph;
FEXCore::IR::Pass* CompactionPass;
bool OptimizeSRA;
bool SupportsAVX;
fextl::vector<LiveRange> LiveRanges;
fextl::unordered_map<IR::NodeID, BlockInterferences> LocalBlockInterferences;
BlockInterferences GlobalBlockInterferences;
[[nodiscard]] static constexpr uint32_t InfoMake(uint32_t id, uint32_t Class) {
return id | (Class << 24);
}
@@ -268,6 +273,8 @@ namespace {
void CalculateLiveRange(FEXCore::IR::IRListView *IR);
void OptimizeStaticRegisters(FEXCore::IR::IRListView *IR);
void CalculateBlockInterferences(FEXCore::IR::IRListView *IR);
void CalculateBlockNodeInterference(FEXCore::IR::IRListView *IR);
void CalculateNodeInterference(FEXCore::IR::IRListView *IR);
void AllocateVirtualRegisters();
void CalculatePredecessors(FEXCore::IR::IRListView *IR);
@@ -288,14 +295,10 @@ namespace {
uint32_t FindSpillSlot(IR::NodeID Node, FEXCore::IR::RegisterClassType RegisterClass);
bool RunAllocateVirtualRegisters(IREmitter *IREmit);
uint64_t OriginalRIP;
fextl::vector<LiveRange*> StaticMaps;
};
ConstrainedRAPass::ConstrainedRAPass(FEXCore::IR::Pass* _CompactionPass, bool _SupportsAVX)
: CompactionPass {_CompactionPass}, SupportsAVX{_SupportsAVX} {
ConstrainedRAPass::ConstrainedRAPass(FEXCore::IR::Pass* _CompactionPass, bool _OptimizeSRA, bool _SupportsAVX)
: CompactionPass {_CompactionPass}, OptimizeSRA(_OptimizeSRA), SupportsAVX{_SupportsAVX} {
}
ConstrainedRAPass::~ConstrainedRAPass() {
@@ -545,7 +548,7 @@ namespace {
auto GprSize = Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount;
auto MapsSize = Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount + Graph->Set.Classes[FPRFixedClass.Val].PhysicalCount;
StaticMaps.resize(MapsSize);
LiveRange* StaticMaps[MapsSize];
// Get a StaticMap entry from context offset
const auto GetStaticMapFromOffset = [&](uint32_t Offset) -> LiveRange** {
@@ -620,7 +623,7 @@ namespace {
// - Mark read-aliases
// - Demote read-aliases if SRA reg is written before the alias's last read
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
memset(StaticMaps.data(), 0, MapsSize * sizeof(LiveRange*));
memset(StaticMaps, 0, MapsSize * sizeof(LiveRange*));
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
const auto Node = IR->GetID(CodeNode);
auto& NodeLiveRange = LiveRanges[Node.Value];
@@ -697,10 +700,8 @@ namespace {
// Marking here as written is overly agressive, but
// there might be write(s) later on the instruction stream
if ((*StaticMap)) {
SRA_DEBUG(
"Marking ssa{} as written because ssa{} re-loads sra{}, "
"and we can't track possible future writes\n",
(*StaticMap) - &LiveRanges[0], Node, -1 /*vreg*/);
SRA_DEBUG("Markng ssa{} as written because ssa{} re-loads sra{}, and we can't track possible future writes\n",
(*StaticMap) - &LiveRanges[0], Node, -1 /*vreg*/);
(*StaticMap)->Written = true;
}
@@ -742,6 +743,103 @@ namespace {
}
}
void ConstrainedRAPass::CalculateBlockInterferences(FEXCore::IR::IRListView *IR) {
using namespace FEXCore;
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
const auto BlockNodeID = IR->GetID(BlockNode);
const auto BlockBeginID = BlockIROp->Begin.ID();
const auto BlockLastID = BlockIROp->Last.ID();
auto& BlockInterferenceVector = LocalBlockInterferences.try_emplace(BlockNodeID).first->second;
BlockInterferenceVector.reserve(BlockLastID.Value - BlockBeginID.Value);
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
const auto Node = IR->GetID(CodeNode);
LiveRange& NodeLiveRange = LiveRanges[Node.Value];
if (NodeLiveRange.Begin >= BlockBeginID &&
NodeLiveRange.End <= BlockLastID) {
// If the live range of this node is FULLY inside of the block
// Then add it to the block specific interference list
BlockInterferenceVector.emplace_back(Node);
}
else {
// If the live range is not fully inside the block then add it to the global interference list
GlobalBlockInterferences.emplace_back(Node);
}
}
}
}
void ConstrainedRAPass::CalculateBlockNodeInterference(FEXCore::IR::IRListView *IR) {
#if 0
const auto AddInterference = [&](IR::NodeID Node1, IR::NodeID Node2) {
RegisterNode *Node = &Graph->Nodes[Node1.Value];
Node->Interference.Set(Node2);
Node->InterferenceList[Node->Head.InterferenceCount++] = Node2;
};
const auto CheckInterferenceNodeSizes = [&](IR::NodeID Node1, uint32_t MaxNewNodes) {
RegisterNode *Node = &Graph->Nodes[Node1.Value];
uint32_t NewListMax = Node->Head.InterferenceCount + MaxNewNodes;
if (Node->InterferenceListSize <= NewListMax) {
const auto AlignedListCount = static_cast<uint32_t>(FEXCore::AlignUp(NewListMax, DEFAULT_INTERFERENCE_LIST_COUNT));
Node->InterferenceListSize = std::max(Node->InterferenceListSize * 2U, AlignedListCount);
Node->InterferenceList = reinterpret_cast<uint32_t*>(realloc(Node->InterferenceList, Node->InterferenceListSize * sizeof(uint32_t)));
}
};
using namespace FEXCore;
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
BlockInterferences *BlockInterferenceVector = &LocalBlockInterferences.try_emplace(IR->GetID(BlockNode)).first->second;
fextl::vector<IR::NodeID> Interferences;
Interferences.reserve(BlockInterferenceVector->size() + GlobalBlockInterferences.size());
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
const auto Node = IR->GetID(CodeNode);
const auto& NodeLiveRange = LiveRanges[Node.Value];
// Check for every interference with the local block's interference
for (auto RHSNode : *BlockInterferenceVector) {
const auto& RHSNodeLiveRange = LiveRanges[RHSNode.Value];
if (!(NodeLiveRange.Begin >= RHSNodeLiveRange.End ||
RHSNodeLiveRange.Begin >= NodeLiveRange.End)) {
Interferences.emplace_back(RHSNode);
}
}
// Now check the global block interference vector
for (auto RHSNode : GlobalBlockInterferences) {
const auto& RHSNodeLiveRange = LiveRanges[RHSNode.Value];
if (!(NodeLiveRange.Begin >= RHSNodeLiveRange.End ||
RHSNodeLiveRange.Begin >= NodeLiveRange.End)) {
Interferences.emplace_back(RHSNode);
}
}
CheckInterferenceNodeSizes(Node, Interferences.size());
for (auto RHSNode : Interferences) {
AddInterference(Node, RHSNode);
}
for (auto RHSNode : Interferences) {
AddInterference(RHSNode, Node);
CheckInterferenceNodeSizes(RHSNode, 0);
}
Interferences.clear();
}
}
#endif
}
void ConstrainedRAPass::CalculateNodeInterference(FEXCore::IR::IRListView *IR) {
const auto AddInterference = [this](IR::NodeID Node1, IR::NodeID Node2) {
RegisterNode *Node = &Graph->Nodes[Node1.Value];
@@ -1127,7 +1225,7 @@ namespace {
if (!CurrentNodes.contains(InterferenceNode)) {
InterferenceIdToSpill = InterferenceNode;
LogMan::Msg::DFmt("[RIP: 0x{:x}] Panic spilling %{}, Live Range[{}, {})", OriginalRIP, InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
LogMan::Msg::DFmt("Panic spilling %{}, Live Range[{}, {})", InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
return true;
}
return false;
@@ -1280,9 +1378,8 @@ namespace {
auto FilledInterference = IREmit->_FillRegister(InterferenceOrderedNode, SpillSlot, InterferenceRegClass);
FilledInterference.first->Header.Size = InterferenceIROp->Size;
FilledInterference.first->Header.ElementSize = InterferenceIROp->ElementSize;
IREmit->ReplaceUsesWithAfter(InterferenceOrderedNode,
FilledInterference,
FilledInterference);
IREmit->ReplaceUsesWithAfter(InterferenceOrderedNode, FilledInterference, FilledInterference);
Spilled = true;
}
}
}
@@ -1295,6 +1392,9 @@ namespace {
using namespace FEXCore;
bool Changed = false;
GlobalBlockInterferences.clear();
LocalBlockInterferences.clear();
auto IR = IREmit->ViewIR();
uint32_t SSACount = IR.GetSSACount();
@@ -1302,8 +1402,18 @@ namespace {
ResetRegisterGraph(Graph, SSACount);
FindNodeClasses(Graph, &IR);
CalculateLiveRange(&IR);
OptimizeStaticRegisters(&IR);
CalculateNodeInterference(&IR);
if (OptimizeSRA)
OptimizeStaticRegisters(&IR);
// Linear forward scan based interference calculation is faster for smaller blocks
// Smarter block based interference calculation is faster for larger blocks
/*if (SSACount >= 2048) {
CalculateBlockInterferences(&IR);
CalculateBlockNodeInterference(&IR);
}
else*/ {
CalculateNodeInterference(&IR);
}
AllocateVirtualRegisters();
return Changed;
@@ -1334,8 +1444,6 @@ namespace {
auto IR = IREmit->ViewIR();
auto HeaderOp = IR.GetHeader();
OriginalRIP = HeaderOp->OriginalRIP;
SpillSlotCount = 0;
Graph->SpillStack.clear();
@@ -1362,7 +1470,7 @@ namespace {
return Changed;
}
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool SupportsAVX) {
return fextl::make_unique<ConstrainedRAPass>(CompactionPass, SupportsAVX);
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool OptimizeSRA, bool SupportsAVX) {
return fextl::make_unique<ConstrainedRAPass>(CompactionPass, OptimizeSRA, SupportsAVX);
}
}
@@ -6,11 +6,11 @@ desc: Sanity Checking
$end_info$
*/
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/set.h>
Loaded 100 of 470 files, more files were not shown because too many files have changed in this diff. Show more