Compare commits

..
6 Commits
Author SHA1 Message Date
Ryan Houdek c094dc238e Docs: Update for release FEX-2509.1 2025-09-15 18:33:36 -07:00
Billy Laws ceaf38e996 Dispatcher: Fix FABI_F32_I16_F80_PTR argument size
This takes an f80 as input and returns an f32. A copy-paste error had
this truncating the input float if !TMP_ABIARGS.
2025-09-15 18:31:55 -07:00
Billy Laws 85e9e255a5 unittests: Add test for x87 mode switches wrongly flushing NZCV 2025-09-15 18:31:50 -07:00
Billy Laws a545865ab7 OpcodeDispatcher: Only flush MMX registers on MMX -> x87 transitions
Flushing other regs is not necessary, and breaks any ConvertNZCVToX87 use
which relies previously saved NZCV values as the flag-setting NZCV op after
the save could trigger a flush of NZCV.
2025-09-15 18:31:44 -07:00
Billy Laws aa8e8f2cb0 OpcodeDispatcher: Don't assert on invalid ALU op encoding 2025-09-15 18:31:37 -07:00
Billy Laws d3a8701e1a WOW64: Fix CsSeg initialization 2025-09-15 18:31:31 -07:00
882 changed files with 61073 additions and 85835 deletions

No files matched your search

-3
View File
@@ -7,6 +7,3 @@ FEXCore/Source/Interface/Core/X86Tables/*
# Inline headers with list-like content that can't be processed individually
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
# Include files in unittests
unittests/*ASM/Includes/*.inc
-2
View File
@@ -20,5 +20,3 @@
# Whole-tree reformat with clang-format-19
5267cde60e7642852d18f20ae8568643bb5293d5
# Minor reformat with clang-format-19
9fdd96af61c969cb5732471223f00eda64b7a069
+1
View File
@@ -34,6 +34,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -41,6 +41,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -34,6 +34,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -33,6 +33,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+2 -1
View File
@@ -48,6 +48,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
@@ -77,7 +78,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
-79
View File
@@ -1,79 +0,0 @@
name: steamrt4 build
on:
push:
branches:
- main
pull_request:
branches:
- main
env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
jobs:
steamrt4_build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARM64, distrobox]]
fail-fast: false
steps:
- uses: actions/checkout@v3
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name : submodule checkout
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: |
rm -Rf ${{runner.workspace}}/build
cmake -E make_directory ${{runner.workspace}}/build
# Setup everything required.
- name : distrobox setup
run: |
distrobox create -Y -i registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306 steamrt4 || true
distrobox upgrade steamrt4
distrobox enter --name steamrt4 -- sudo apt-get install -y \
git cmake ninja-build ccache \
lld clang \
libclang-dev llvm-dev \
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross \
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
- name: Create Build Environment
run: distrobox enter --name steamrt4 -- cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
shell: bash
working-directory: ${{runner.workspace}}/build
run: distrobox enter --name steamrt4 -- cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr
- name: Build
working-directory: ${{runner.workspace}}/build
shell: bash
run: distrobox enter --name steamrt4 -- cmake --build . --config $BUILD_TYPE
- name: install
working-directory: ${{runner.workspace}}/build
shell: bash
env:
DESTDIR: ${{runner.workspace}}/install
run: distrobox enter --name steamrt4 -- cmake --build . --config $BUILD_TYPE -t install
- name: Upload libraries
uses: 'actions/upload-artifact@v4'
timeout-minutes: 1
with:
overwrite: true
name: steamrt4_steampipe_depot
path: ${{runner.workspace}}/install/*
retention-days: 60
compression-level: 9
+1
View File
@@ -35,6 +35,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+2 -2
View File
@@ -46,12 +46,12 @@ jobs:
- name: Configure CMake arm64ec
shell: bash
working-directory: ${{runner.workspace}}/build_arm64ec
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
- name: Configure CMake wow64
shell: bash
working-directory: ${{runner.workspace}}/build_wow64
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
- name: Build arm64ec
working-directory: ${{runner.workspace}}/build_arm64ec
+197 -193
View File
@@ -1,49 +1,47 @@
cmake_minimum_required(VERSION 3.14)
project(FEX C CXX ASM)
include(CheckIncludeFiles)
check_include_files("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
INCLUDE (CheckIncludeFiles)
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests (requires x86 compiler)" FALSE)
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
option(BUILD_THUNKS "Build thunks" FALSE)
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
option(ENABLE_CLANG_THUNKS "Build thunks with clang" TRUE)
option(ENABLE_IWYU "Enable the Include What You Use sanitizer" FALSE)
option(ENABLE_IWYU "Enables include what you use program" FALSE)
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
set(USE_LINKER "" CACHE STRING "Path to a custom linker program")
option(ENABLE_UBSAN "Enable the Clang Undefined Behavior Sanitizer" FALSE)
option(ENABLE_ASAN "Enable the Clang Address Sanitizer" FALSE)
option(ENABLE_TSAN "Enable the Clang Thread Sanitizer" FALSE)
option(ENABLE_COVERAGE "Enable Code Coverage" FALSE)
option(ENABLE_ASSERTIONS "Enable debug assertions" FALSE)
option(ENABLE_GDB_SYMBOLS "Enable GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
option(ENABLE_STRICT_WERROR "Enable stricter -Werror" FALSE)
option(ENABLE_WERROR "Enable -Werror" FALSE)
option(ENABLE_JEMALLOC "Enable jemalloc allocator" TRUE)
option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enable jemalloc glibc allocator" TRUE)
option(ENABLE_OFFLINE_TELEMETRY "Enable FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enable time trace compile option" FALSE)
option(ENABLE_LIBCXX "Use LLVM's libc++ instead of the GNU libstdc++" FALSE)
option(ENABLE_CCACHE "Enable ccache for build caching" TRUE)
option(ENABLE_VIXL_SIMULATOR "Use the VIXL simulator for emulation (only useful for CI testing)" FALSE)
option(ENABLE_VIXL_DISASSEMBLER "Enable debug disassembler output with VIXL" FALSE)
option(USE_LEGACY_BINFMTMISC "Use legacy method of setting up binfmt_misc" FALSE)
option(ENABLE_FEXCORE_PROFILER "Enable FEXCore's timeline profiling capabilities" FALSE)
set(FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for FEXCore's profiler")
set_property(CACHE FEXCORE_PROFILER_BACKEND PROPERTY STRINGS gpuvis tracy)
set(USE_LINKER "" CACHE STRING "Allow overriding the linker path directly")
option(ENABLE_UBSAN "Enables Clang UBSAN" FALSE)
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
option(ENABLE_COVERAGE "Enables Coverage" FALSE)
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
option(ENABLE_WERROR "Enables -Werror" FALSE)
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enables jemalloc glibc allocator" TRUE)
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_VIXL_SIMULATOR "Enable use of VIXL simulator for emulation (only useful for CI testing)" FALSE)
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for the FEXCore profiler (gpuvis, tracy)")
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
option(USE_PDB_DEBUGINFO "Build debug info in PDB format" FALSE)
option(BUILD_STEAM_SUPPORT "Enable Steam integration" FALSE)
set(X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
set(X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
set(X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
set(DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
set(HOSTLIBS_DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
set (X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
set (DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
set (HOSTLIBS_DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
if (NOT DATA_DIRECTORY)
set(DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
endif()
include(GNUInstallDirs)
@@ -51,74 +49,22 @@ if (NOT HOSTLIBS_DATA_DIRECTORY)
set(HOSTLIBS_DATA_DIRECTORY "${CMAKE_INSTALL_FULL_LIBDIR}/fex-emu")
endif()
## Platform Checks ##
# Only 64-bit Linux and Windows are supported
# NB: SIZEOF_VOID_P is in bytes, not bits
# On 32-bit systems this is set to 4
if (NOT CMAKE_SIZEOF_VOID_P EQUAL 8)
message(FATAL_ERROR "Unsupported pointer size ${CMAKE_SIZEOF_VOID_P}."
" FEX only supports 64-bit (8-byte pointer) systems."
" If you believe this is in error, file an issue.")
elseif (NOT (WIN32 OR CMAKE_SYSTEM_NAME STREQUAL "Linux"))
message(FATAL_ERROR "Unsupported system type ${CMAKE_SYSTEM_NAME}."
" FEX only supports Linux and Windows."
" If you believe this is in error, file an issue.")
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
if (NOT CONTAINS_MINGW EQUAL -1)
message (STATUS "Mingw build")
set (MINGW_BUILD TRUE)
set (ENABLE_JEMALLOC TRUE)
set (ENABLE_JEMALLOC_GLIBC_ALLOC FALSE)
endif()
## Compiler Checks ##
# GCC and MSVC are unsupported
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
message(FATAL_ERROR "FEX doesn't support GCC! Use Clang instead.")
elseif (MSVC)
message(FATAL_ERROR "FEX doesn't support MSVC! Use Clang on MinGW instead.")
elseif (MINGW)
message(STATUS "Building for MinGW")
set(ENABLE_JEMALLOC TRUE)
set(ENABLE_JEMALLOC_GLIBC_ALLOC FALSE)
else ()
message(STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
set(CLANG_MINIMUM_VERSION 13.0)
if (NOT MINGW_BUILD)
message (STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
set (CLANG_MINIMUM_VERSION 13.0)
if (CMAKE_CXX_COMPILER_VERSION VERSION_LESS ${CLANG_MINIMUM_VERSION})
message(FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
message (FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
endif()
endif()
## Architecture Handling ##
string(TOLOWER ${CMAKE_SYSTEM_PROCESSOR} processor)
if (processor MATCHES "x86|amd64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
message(FATAL_ERROR
" FEX doesn't support compiling for x86-64 hosts!"
" This is /only/ a supported configuration for FEX CI and nothing else!")
else()
message(STATUS "x86_64 debug build")
endif()
set(ARCHITECTURE_x86_64 1)
add_definitions(-DARCHITECTURE_x86_64=1)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
elseif (processor MATCHES "^aarch64|^arm64|^armv8\.*")
set(ARCHITECTURE_arm64 1)
add_definitions(-DARCHITECTURE_arm64=1)
# arm64ec needs to define both arm64 and arm64ec
if (processor MATCHES "^arm64ec")
set(ARCHITECTURE_arm64ec 1)
add_definitions(-DARCHITECTURE_arm64ec=1)
endif()
endif()
if (NOT (ARCHITECTURE_arm64 OR ARCHITECTURE_arm64ec OR ARCHITECTURE_x86_64))
message(FATAL_ERROR "Unsupported processor type ${processor}."
" If you believe this is in error, file an issue.")
endif()
if (BUILD_STEAM_SUPPORT)
add_definitions(-DFEX_STEAM_SUPPORT=1)
endif()
if (ENABLE_FEXCORE_PROFILER)
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
@@ -138,8 +84,8 @@ if (ENABLE_FEXCORE_PROFILER)
add_definitions(-DTRACY_NO_SAMPLING=1)
# This pulls in libbacktrace which allocators in global constructors (before FEX can set up its allocator hooks)
add_definitions(-DTRACY_NO_CALLSTACK=1)
if (MINGW)
message(FATAL_ERROR "Tracy profiler not supported on MinGW")
if (MINGW_BUILD)
message(FATAL_ERROR "Tracy profiler not supported")
endif()
else()
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
@@ -166,10 +112,9 @@ if(NOT TARGET uninstall)
endif()
# These options are meant for package management
set(TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
set(TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
set(OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version")
set(OVERRIDE_HASH "detect" CACHE STRING "Override the FEX git hash")
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
set (OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version in the format of <MMYY>{.<REV>}")
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
if (CMAKE_BUILD_TYPE MATCHES "DEBUG")
@@ -186,6 +131,7 @@ if (ENABLE_GDB_SYMBOLS)
add_definitions(-DGDB_SYMBOLS_ENABLED=1)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
@@ -195,7 +141,33 @@ cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
include(CheckPIESupported)
check_pie_supported()
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION ${ENABLE_LTO})
if (ENABLE_LTO)
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION TRUE)
else()
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION FALSE)
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
message(FATAL_ERROR
" FEX-Emu doesn't support compiling for x86-64 hosts!"
" This is /only/ a supported configuration for FEX CI and nothing else!")
endif()
set(_M_X86_64 1)
add_definitions(-D_M_X86_64=1)
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
set(_M_ARM_64 1)
add_definitions(-D_M_ARM_64=1)
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
set(_M_ARM_64EC 1)
add_definitions(-D_M_ARM_64EC=1)
endif()
include(CheckCXXSourceCompiles)
set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
@@ -211,15 +183,15 @@ check_cxx_source_compiles(
HAS_CLANG_PRESERVE_ALL)
unset(CMAKE_REQUIRED_FLAGS)
if (HAS_CLANG_PRESERVE_ALL)
if (MINGW)
if (MINGW_BUILD)
message(STATUS "Ignoring broken clang::preserve_all support")
set(HAS_CLANG_PRESERVE_ALL FALSE)
else()
message(STATUS "Has clang::preserve_all")
endif()
endif()
endif ()
if (ARCHITECTURE_arm64 AND HAS_CLANG_PRESERVE_ALL)
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
add_definitions("-DFEX_PRESERVE_ALL_ATTR=__attribute__((preserve_all))" "-DFEX_HAS_PRESERVE_ALL_ATTR=1")
else()
add_definitions("-DFEX_PRESERVE_ALL_ATTR=" "-DFEX_HAS_PRESERVE_ALL_ATTR=0")
@@ -248,7 +220,7 @@ if (ENABLE_COMPILE_TIME_TRACE)
link_libraries(-ftime-trace)
endif()
set(PTHREAD_LIB pthread)
set (PTHREAD_LIB pthread)
if (USE_LINKER)
message(STATUS "Overriding linker to: ${USE_LINKER}")
@@ -300,8 +272,8 @@ if (ENABLE_JEMALLOC_GLIBC_ALLOC)
# Required for thunks to work.
# All host native libraries will use this allocator, while *most* other FEX internal allocations will use the other jemalloc allocator.
add_subdirectory(External/jemalloc_glibc/)
elseif (NOT MINGW)
message(STATUS
elseif (NOT MINGW_BUILD)
message (STATUS
" jemalloc glibc allocator disabled!\n"
" This is not a recommended configuration!\n"
" This will very explicitly break thunk execution!\n"
@@ -311,8 +283,8 @@ endif()
if (ENABLE_JEMALLOC)
# The jemalloc subproject that all FEXCore fextl objects allocate through.
add_subdirectory(External/jemalloc/)
elseif (NOT MINGW)
message(STATUS
elseif (NOT MINGW_BUILD)
message (STATUS
" jemalloc disabled!\n"
" This is not a recommended configuration!\n"
" This will very explicitly break 32-bit application execution!\n"
@@ -324,22 +296,15 @@ if (USE_PDB_DEBUGINFO)
add_link_options(-g -Wl,--pdb=)
endif()
set(CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set(CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set(CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set(CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
## Modules ##
list(APPEND CMAKE_MODULE_PATH ${CMAKE_SOURCE_DIR}/Data/CMake/)
include(LinkerGC)
## Externals ##
include_directories(External/robin-map/include/)
include(CTest)
if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
add_subdirectory(External/vixl/)
include_directories(SYSTEM External/vixl/src/)
endif()
@@ -348,16 +313,20 @@ if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
add_subdirectory(External/tracy)
endif()
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
# This means we were attempted to get compiled with GCC
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
set(BUILD_SHARED_LIBS OFF)
if (NOT CMAKE_CROSSCOMPILING)
find_package(xxhash MODULE QUIET)
endif()
if (NOT TARGET xxHash::xxhash)
pkg_search_module(xxhash IMPORTED_TARGET xxhash libxxhash)
if (TARGET PkgConfig::xxhash AND NOT CMAKE_CROSSCOMPILING)
add_library(xxHash::xxhash ALIAS PkgConfig::xxhash)
else()
set(XXHASH_BUNDLED_MODE TRUE)
set(XXHASH_BUILD_XXHSUM FALSE)
add_subdirectory(External/xxhash/cmake_unofficial/)
@@ -366,7 +335,7 @@ endif()
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
if (BUILD_TESTING)
if (BUILD_TESTS)
find_package(Catch2 3 QUIET)
if (NOT Catch2_FOUND)
add_subdirectory(External/Catch2/)
@@ -376,9 +345,6 @@ if (BUILD_TESTING)
endif()
include(Catch)
else ()
# Override any previously generated test list to avoid running stale test binaries
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
endif()
find_package(fmt QUIET)
@@ -438,7 +404,7 @@ if (NOT TUNE_ARCH STREQUAL "generic")
endif()
if (TUNE_CPU STREQUAL "native")
if(ARCHITECTURE_arm64)
if(_M_ARM_64)
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
@@ -479,40 +445,6 @@ elseif (NOT TUNE_CPU STREQUAL "none")
endif()
endif()
set(GIT_DESCRIBE_STRING "FEX-Unknown")
if (OVERRIDE_VERSION STREQUAL "detect")
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE)
endif()
else()
set(GIT_DESCRIBE_STRING "${OVERRIDE_VERSION}")
endif()
set(GIT_SHORT_HASH "Unknown")
if (OVERRIDE_HASH STREQUAL "detect")
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} rev-parse --short=7 HEAD
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_SHORT_HASH
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE)
endif()
else()
set(GIT_SHORT_HASH "${OVERRIDE_HASH}")
endif()
if (ENABLE_IWYU)
find_program(IWYU_EXE "iwyu")
if (IWYU_EXE)
@@ -523,10 +455,15 @@ endif()
add_compile_options(-Wall)
if (BUILD_TESTING)
include(CTest)
if (BUILD_TESTS)
message(STATUS "Unit tests are enabled")
if (NOT BUILD_TESTING)
# CMake checks this variable before generating CTestTestfile.cmake
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
endif()
set(TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
if (TEST_JOB_COUNT)
message(STATUS "Running tests with ${TEST_JOB_COUNT} jobs")
elseif(CMAKE_VERSION VERSION_LESS "3.29")
@@ -541,16 +478,13 @@ add_subdirectory(FEXHeaderUtils/)
add_subdirectory(CodeEmitter/)
add_subdirectory(FEXCore/)
if (ARCHITECTURE_arm64 AND NOT MINGW AND NOT BUILD_STEAM_SUPPORT)
if (_M_ARM_64 AND NOT MINGW_BUILD)
# Binfmt_misc files must be installed prior to Source/ installs
add_subdirectory(Data/binfmts/)
endif()
add_subdirectory(Source/)
if (NOT BUILD_STEAM_SUPPORT)
add_subdirectory(Data/AppConfig/)
endif()
add_subdirectory(Data/AppConfig/)
# Install the ThunksDB file
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
@@ -558,16 +492,15 @@ file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.js
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/)
endforeach()
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
if (BUILD_THUNKS)
set(FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
set (FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
add_subdirectory(ThunkLibs/Generator)
# Thunk targets for both host libraries and IDE integration
@@ -594,7 +527,8 @@ if (BUILD_THUNKS)
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
DEPENDS thunkgen)
DEPENDS thunkgen
)
ExternalProject_Add(guest-libs-32
PREFIX guest-libs-32
@@ -612,36 +546,106 @@ if (BUILD_THUNKS)
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
DEPENDS thunkgen)
DEPENDS thunkgen
)
install(
CODE "message(\"-- Installing: guest-libs\")"
CODE "MESSAGE(\"-- Installing: guest-libs\")"
CODE "
execute_process(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest)"
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)"
DEPENDS guest-libs
COMPONENT Runtime)
)
install(
CODE "message(\"-- Installing: guest-libs-32\")"
CODE "MESSAGE(\"-- Installing: guest-libs-32\")"
CODE "
execute_process(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32)"
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)"
DEPENDS guest-libs-32
COMPONENT Runtime)
)
add_custom_target(uninstall_guest-libs
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)
add_custom_target(uninstall_guest-libs-32
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)
add_dependencies(uninstall uninstall_guest-libs)
add_dependencies(uninstall uninstall_guest-libs-32)
endif()
if (BUILD_STEAM_SUPPORT)
add_subdirectory(Source/Steam/)
set(FEX_VERSION_MAJOR "0")
set(FEX_VERSION_MINOR "0")
set(FEX_VERSION_PATCH "0")
if (OVERRIDE_VERSION STREQUAL "detect")
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
RESULT_VARIABLE GIT_ERROR
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
if (NOT ${GIT_ERROR} EQUAL 0)
# Likely built in a way that doesn't have tags
# Setup a version tag that is unknown
set(GIT_DESCRIBE_STRING "FEX-0000")
endif()
endif()
else()
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
endif()
# Parse the version here
# Change something like `FEX-2106.1-76-<hash>` in to a list
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
# Extract the `2106.1` element
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
# Change `2106.1` in to a list
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
# Calculate list size
list(LENGTH DESCRIBE_LIST LIST_SIZE)
# Pull out the major version
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
# Minor version only exists if there is a .1 at the end
# eg: 2106 versus 2106.1
if (LIST_SIZE GREATER 1)
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
endif()
# Package creation
set (CPACK_GENERATOR "DEB")
set (CPACK_PACKAGE_NAME fex-emu)
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/Description.txt")
# Debian defines
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/triggers")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
# binfmt_misc conflicts with qemu-user-static
# We also only install binfmt_misc on aarch64 hosts
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
endif()
include (CPack)
+1 -1
View File
@@ -129,4 +129,4 @@
"variables": []
}
]
}
}
+37 -63
View File
@@ -36,31 +36,24 @@ public:
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
void adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (IsADRRange(Imm)) {
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
return BranchEncodeSucceeded::Success;
}
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
// Can't encode.
return BranchEncodeSucceeded::Failure;
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, ForwardLabel* Label) {
void adr(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADR});
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return adr(rd, &Label->Backward);
adr(rd, &Label->Backward);
} else {
return adr(rd, &Label->Forward);
adr(rd, &Label->Forward);
}
}
@@ -69,53 +62,38 @@ public:
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
void adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) {
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
void adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADRP});
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return adrp(rd, &Label->Backward);
adrp(rd, &Label->Backward);
} else {
return adrp(rd, &Label->Forward);
adrp(rd, &Label->Forward);
}
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
const auto SLocation = reinterpret_cast<int64_t>(Label->Location);
const auto ULocation = std::bit_cast<uint64_t>(SLocation);
const int64_t Imm = SLocation - (GetCursorAddress<int64_t>());
const auto UImm = std::bit_cast<uint64_t>(Imm);
void LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
if (IsADRRange(Imm)) {
// If the range is in ADR range then we can just use ADR.
return adr(rd, Label);
}
if (IsADRPRange(Imm)) {
const int64_t ADRPImm = (SLocation & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
adr(rd, Label);
} else if (IsADRPRange(Imm)) {
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
// If the range is in the ADRP range then we can use ADRP.
const bool NeedsOffset = !IsADRPAligned(ULocation);
const uint64_t AlignedOffset = ULocation & 0xFFFULL;
bool NeedsOffset = !IsADRPAligned(reinterpret_cast<uint64_t>(Label->Location));
uint64_t AlignedOffset = reinterpret_cast<uint64_t>(Label->Location) & 0xFFFULL;
// First emit ADRP
adrp(rd, ADRPImm >> 12);
@@ -124,33 +102,23 @@ public:
// Now even an add
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
}
return BranchEncodeSucceeded::Success;
} else {
LOGMAN_MSG_A_FMT("Unscaled offset too large");
FEX_UNREACHABLE;
}
// Stinky path, we need to load the address as a sequence of movz+movk+movk
movz(ARMEmitter::Size::i64Bit, rd, (UImm >> 32) & 0xFFFF, 32);
movk(ARMEmitter::Size::i64Bit, rd, (UImm >> 16) & 0xFFFF, 16);
movk(ARMEmitter::Size::i64Bit, rd, UImm & 0xFFFF);
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
// Emit a register index and two nops. These will be backpatched.
// Emit a register index and a nop. These will be backpatched.
dc32(rd.Idx());
nop();
nop();
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return LongAddressGen(rd, &Label->Backward);
LongAddressGen(rd, &Label->Backward);
} else {
return LongAddressGen(rd, &Label->Forward);
LongAddressGen(rd, &Label->Forward);
}
}
@@ -894,6 +862,12 @@ public:
}
private:
static constexpr Condition InvertCondition(Condition cond) {
// These behave as always, so it makes no sense to allow inverting these.
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
}
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b001'0010'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
+2 -1
View File
@@ -2244,7 +2244,8 @@ public:
template<IsQOrDRegister T>
void movi(SubRegSize size, T rd, uint64_t Imm, uint16_t Shift = 0) {
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit ||
size == SubRegSize::i64Bit,
"Unsupported movi size");
uint32_t cmode;
+64 -123
View File
@@ -20,31 +20,23 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm);
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
void b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
void b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
void b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return b(Cond, &Label->Backward);
b(Cond, &Label->Backward);
} else {
return b(Cond, &Label->Forward);
b(Cond, &Label->Forward);
}
}
@@ -53,32 +45,24 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm);
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
void bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
void bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return bc(Cond, &Label->Backward);
bc(Cond, &Label->Backward);
} else {
return bc(Cond, &Label->Forward);
bc(Cond, &Label->Forward);
}
}
@@ -114,32 +98,25 @@ public:
UnconditionalBranch(Op, Imm);
}
[[nodiscard]] BranchEncodeSucceeded b(const BackwardLabel* Label) {
void b(const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0001'01 << 26;
// Can't encode.
return BranchEncodeSucceeded::Failure;
UnconditionalBranch(Op, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded b(ForwardLabel* Label) {
void b(ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded b(BiDirectionalLabel* Label) {
void b(BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return b(&Label->Backward);
b(&Label->Backward);
} else {
return b(&Label->Forward);
b(&Label->Forward);
}
}
@@ -149,33 +126,25 @@ public:
UnconditionalBranch(Op, Imm);
}
[[nodiscard]] BranchEncodeSucceeded bl(const BackwardLabel* Label) {
void bl(const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b1001'01 << 26;
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
UnconditionalBranch(Op, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded bl(ForwardLabel* Label) {
void bl(ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded bl(BiDirectionalLabel* Label) {
void bl(BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return bl(&Label->Backward);
bl(&Label->Backward);
} else {
return bl(&Label->Forward);
bl(&Label->Forward);
}
}
@@ -186,35 +155,28 @@ public:
CompareAndBranch(Op, s, rt, Imm);
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0100 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return cbz(s, rt, &Label->Backward);
cbz(s, rt, &Label->Backward);
} else {
return cbz(s, rt, &Label->Forward);
cbz(s, rt, &Label->Forward);
}
}
@@ -224,35 +186,28 @@ public:
CompareAndBranch(Op, s, rt, Imm);
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0101 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return cbnz(s, rt, &Label->Backward);
cbnz(s, rt, &Label->Backward);
} else {
return cbnz(s, rt, &Label->Forward);
cbnz(s, rt, &Label->Forward);
}
}
@@ -262,35 +217,28 @@ public:
TestAndBranch(Op, rt, Bit, Imm);
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0110 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return tbz(rt, Bit, &Label->Backward);
tbz(rt, Bit, &Label->Backward);
} else {
return tbz(rt, Bit, &Label->Forward);
tbz(rt, Bit, &Label->Forward);
}
}
@@ -299,34 +247,27 @@ public:
TestAndBranch(Op, rt, Bit, Imm);
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0111 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return tbnz(rt, Bit, &Label->Backward);
tbnz(rt, Bit, &Label->Backward);
} else {
return tbnz(rt, Bit, &Label->Forward);
tbnz(rt, Bit, &Label->Forward);
}
}
+31 -78
View File
@@ -586,15 +586,6 @@ concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegi
template<typename T>
concept IsQOrDRegister = std::is_same_v<T, QRegister> || std::is_same_v<T, DRegister>;
template<typename T>
concept IsLabel = std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>;
enum class BranchEncodeSucceeded {
Success,
Failure,
};
// Whether or not a given set of vector registers are sequential
// in increasing order as far as the register file is concerned (modulo its size)
//
@@ -647,25 +638,19 @@ public:
// Bind a backward label to an address.
// Address that is bound is the current emitter location.
[[nodiscard]] bool Bind(BackwardLabel* Label) {
void Bind(BackwardLabel* Label) {
LOGMAN_THROW_A_FMT(Label->Location == nullptr, "Trying to bind a label twice");
Label->Location = GetCursorAddress<uint8_t*>();
// Always binds because it is only storing a location.
return true;
}
[[nodiscard]] bool Bind(const ForwardLabel::Reference* Label) {
void Bind(const ForwardLabel::Reference* Label) {
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
// Patch up the instructions
switch (Label->Type) {
case ForwardLabel::InstType::ADR: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!IsADRRange(Imm)) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
@@ -677,12 +662,7 @@ public:
case ForwardLabel::InstType::ADRP: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
@@ -692,13 +672,11 @@ public:
*Instruction = Inst;
break;
}
case ForwardLabel::InstType::B: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -708,13 +686,11 @@ public:
break;
}
case ForwardLabel::InstType::TEST_BRANCH: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -728,10 +704,7 @@ public:
case ForwardLabel::InstType::RELATIVE_LOAD: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -741,44 +714,38 @@ public:
break;
}
case ForwardLabel::InstType::LONG_ADDRESS_GEN: {
const auto* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
const auto ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
const auto ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
const auto ImmInstThree = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[2]);
const auto OriginalOffset = GetCursorOffset();
uint32_t* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
auto OriginalOffset = GetCursorOffset();
const auto InstOffset = GetCursorOffsetFromAddress(Instructions);
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
SetCursorOffset(InstOffset);
// We encoded the destination register in to the first instruction space.
// Read it back.
ARMEmitter::Register DestReg(Instructions[0]);
if (IsADRRange(ImmInstThree)) {
// If within ADR range from the third instruction, then we can emit NOP+NOP+ADR
if (IsADRRange(ImmInstTwo)) {
// If within ADR range from the second instruction, then we can emit NOP+ADR
nop();
nop();
adr(DestReg, static_cast<uint32_t>(ImmInstThree) & 0x7FFF);
} else if (IsADRPRange(ImmInstTwo)) {
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
} else if (IsADRPRange(ImmInstOne)) {
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
// First check if we are in non-offset range for second instruction.
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
// We can emit nop + nop + adrp
nop();
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstThree >> 12) & 0x7FFF);
} else {
// Not aligned, need nop + adrp + add
// We can emit nop + adrp
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstTwo & 0xFFF);
} else {
// Not aligned, need adrp + add
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
}
} else {
// Stinky path, we need to emit a movz+movk+movk sequence.
movz(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 32) & 0x7FFF, 32);
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 16) & 0xFFFF, 16);
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne) & 0xFFFF);
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
FEX_UNREACHABLE;
}
SetCursorOffset(OriginalOffset);
@@ -786,41 +753,27 @@ public:
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
return true;
}
// Bind a forward label to a location.
// This walks all the instructions in the label's vector.
// Then backpatching all instructions that have used the label.
[[nodiscard]] bool Bind(ForwardLabel* Label) {
bool Bound = true;
void Bind(ForwardLabel* Label) {
if (Label->FirstInst.Location) {
Bound &= Bind(&Label->FirstInst);
Bind(&Label->FirstInst);
}
for (auto& Inst : Label->Insts) {
Bound &= Bind(&Inst);
Bind(&Inst);
}
return Bound;
}
// Bind a bidirectional location to a location.
// Binds both forwards and backwards depending on how the label was used.
[[nodiscard]] bool Bind(BiDirectionalLabel* Label) {
bool Bound = true;
void Bind(BiDirectionalLabel* Label) {
if (!Label->Backward.Location) {
Bound &= Bind(&Label->Backward);
Bind(&Label->Backward);
}
Bound &= Bind(&Label->Forward);
return Bound;
}
static constexpr Condition InvertCondition(Condition cond) {
// These behave as always, so it makes no sense to allow inverting these.
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
Bind(&Label->Forward);
}
#include <CodeEmitter/VixlUtils.inl>
+3 -3
View File
@@ -5125,7 +5125,7 @@ private:
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
[[nodiscard]]
static bool IsValidFPValueForImm8(T value) {
const uint64_t bits = std::bit_cast<FloatToEquivalentUInt<T>>(value);
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
static constexpr std::array mantissa_masks {
@@ -5171,7 +5171,7 @@ protected:
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
#endif
const auto bits = std::bit_cast<uint32_t>(value);
const auto bits = FEXCore::BitCast<uint32_t>(value);
const auto sign = (bits & 0x80000000) >> 24;
const auto expb2 = (bits & 0x20000000) >> 23;
const auto b5_to_0 = (bits >> 19) & 0x3F;
@@ -5184,7 +5184,7 @@ protected:
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
#endif
const auto bits = std::bit_cast<uint64_t>(value);
const auto bits = FEXCore::BitCast<uint64_t>(value);
const auto sign = (bits & 0x80000000'00000000) >> 56;
const auto expb2 = (bits & 0x20000000'00000000) >> 55;
const auto b5_to_0 = (bits >> 48) & 0x3F;
+7 -6
View File
@@ -4,8 +4,7 @@ file(GLOB GEN_CONFIG_SOURCES CONFIGURE_DEPENDS *.json.in)
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
# Any configuration file json file that needs to be generated
@@ -15,10 +14,12 @@ foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WLE)
# Configure it
configure_file(${GEN_CONFIG_SRC} ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
configure_file(
${GEN_CONFIG_SRC}
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
# Then install the configured json
install(FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
install(
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
+3
View File
@@ -0,0 +1,3 @@
x86 and x86-64 Linux emulator
FEX allows you to run x86 applications on ARM64 Linux devices. It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
+18
View File
@@ -0,0 +1,18 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Setup binfmt_misc
update-binfmts --import FEX-x86
update-binfmts --import FEX-x86_64
}
# Install FEXInterpreter hardlink
# Needs to be done before setting up binfmt_misc
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
+17
View File
@@ -0,0 +1,17 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Uninstall
update-binfmts --unimport FEX-x86
update-binfmts --unimport FEX-x86_64
}
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
# Remove FEXInterpreter hardlink
unlink /usr/bin/FEXInterpreter
+1
View File
@@ -0,0 +1 @@
activate-noawait ldconfig
-18
View File
@@ -1,18 +0,0 @@
# SPDX-License-Identifier: MIT
include(FindPackageHandleStandardArgs)
find_package(PkgConfig QUIET)
pkg_search_module(xxhash QUIET IMPORTED_TARGET xxhash libxxhash)
find_package_handle_standard_args(xxhash
REQUIRED_VARS xxhash_LINK_LIBRARIES
VERSION_VAR xxhash_VERSION
)
if (xxhash_FOUND AND NOT TARGET xxHash::xxhash)
if (TARGET PkgConfig::xxhash)
add_library(xxHash::xxhash ALIAS PkgConfig::xxhash)
else()
add_library(xxHash::xxhash ALIAS xxhash)
endif()
endif()
-15
View File
@@ -1,15 +0,0 @@
# SPDX-License-Identifier: MIT
# This applies some common linker options that reduce code size and linking time in Release mode. Namely:
# --gc-sections: Linktime garbage collection, discards unused sections from the final output
# --strip-all : Similar to running `strip`, discards the symbol table from the final output
# --as-needed : Only includes libraries that are actually needed in the final output.
macro(LinkerGC target)
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
target_link_options(${target} PRIVATE
"LINKER:--gc-sections"
"LINKER:--strip-all"
"LINKER:--as-needed")
endif()
endmacro()
+1 -1
View File
@@ -14,7 +14,7 @@ RUN mkdir build
ARG CC=clang-13
ARG CXX=clang++-13
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTING=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTS=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN ninja
WORKDIR /FEX/build
+9 -7
View File
@@ -3,21 +3,23 @@ function(GenBinFmt Name)
get_filename_component(FMT_NAME ${Name} NAME_WE)
# Configure it
configure_file(${Name} ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
configure_file(
${Name}
${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
# Then install the configured binfmt
install(FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/
COMPONENT Runtime)
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
endfunction()
if (NOT USE_LEGACY_BINFMTMISC)
configure_file(FEX-x86.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf)
configure_file(FEX-x86_64.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf)
install(FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/
COMPONENT Runtime)
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
else()
GenBinFmt(FEX-x86.in)
GenBinFmt(FEX-x86_64.in)
+1 -1
View File
@@ -1 +1 @@
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+1 -1
View File
@@ -1 +1 @@
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+3 -3
View File
@@ -2,8 +2,8 @@
let
toolchain = pkgs.fetchzip {
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250920/llvm-mingw-20250920-ucrt-ubuntu-22.04-aarch64.tar.xz";
sha256 = "sha256-LaojKjC8KzY+soW5u6eoDoXE3qtYk9Ejr7M3enTqRAE=";
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250305/llvm-mingw-20250305-ucrt-ubuntu-20.04-aarch64.tar.xz";
sha256 = "sha256-cA03/ab9O61eO9+S2JzIXD4V0HzTXK5/AYyxW2d73Po=";
};
cmakeToolchainFile = pkgs.substitute {
@@ -45,7 +45,7 @@ pkgs.mkShell {
fi
'';
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False
FEX_CMAKE_TOOLCHAIN_ARM64EC = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_CMAKE_TOOLCHAIN_WOW64 = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_MESON_CROSSFILE = "--cross-file ${mesonCrossFile}";
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -14,4 +14,4 @@ fi
rm -rf unittests/FEXLinuxTests
set -o xtrace
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTING=ON -DBUILD_FEX_LINUX_TESTS=ON
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTS=ON -DBUILD_FEX_LINUX_TESTS=ON
+1 -1
+3 -2
View File
@@ -1,5 +1,5 @@
add_library(softfloat_3e STATIC
set (SRCS
# F80 support
src/extF80_add.c
src/extF80_div.c
@@ -84,7 +84,7 @@ add_library(softfloat_3e STATIC
src/s_normSubnormalF32Sig.c
src/s_f32UIToCommonNaN.c)
if (ARCHITECTURE_arm64 AND HAS_CLANG_PRESERVE_ALL)
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=__attribute__((preserve_all));-DFEXCORE_HAS_PRESERVE_ALL_ATTR=1")
else()
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=;-DFEXCORE_HAS_PRESERVE_ALL_ATTR=0")
@@ -92,6 +92,7 @@ endif()
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ=1;-DINLINE=static inline;-DINLINE_LEVEL=4;-DSOFTFLOAT_FAST_INT64=1;-DSOFTFLOAT_FAST_DIV32TO16=1;-DSOFTFLOAT_FAST_DIV64TO32=1")
add_library(softfloat_3e STATIC ${SRCS})
target_include_directories(softfloat_3e PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include/)
target_include_directories(softfloat_3e PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include/SoftFloat-3e/)
target_compile_definitions(softfloat_3e PUBLIC ${DEFINES})
+2 -1
View File
@@ -1,4 +1,4 @@
add_library(cephes_128bit STATIC
set(SRCS_128BIT
src/128bit/Impl.cpp
src/128bit/atanll.c
src/128bit/constll.c
@@ -11,6 +11,7 @@ add_library(cephes_128bit STATIC
src/128bit/tanll.c)
# 128-bit library
add_library(cephes_128bit STATIC ${SRCS_128BIT})
target_link_libraries(cephes_128bit softfloat_3e)
target_include_directories(cephes_128bit PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include/)
target_compile_options(cephes_128bit PRIVATE -fno-builtin)
+3 -3
View File
@@ -306,9 +306,9 @@ typing-extensions==4.14.1 \
--hash=sha256:38b39f4aeeab64884ce9f74c94263ef78f3c22467c8724005483154c26648d36 \
--hash=sha256:d1e1e3b58374dc93031d6eda2420a48ea44a36c2b4766a4fdeb3710755731d76
# via pygithub
urllib3==2.6.0 \
--hash=sha256:c90f7a39f716c572c4e3e58509581ebd83f9b59cced005b7db7ad2d22b0db99f \
--hash=sha256:cb9bcef5a4b345d5da5d145dc3e30834f58e8018828cbc724d30b4cb7d4d49f1
urllib3==2.5.0 \
--hash=sha256:3fc47733c7e419d4bc3f6b3dc2b4f890bb743906a30d56ba4a5bfa4bbff92760 \
--hash=sha256:e6b01673c0fa6a13e374b50871808eb3bf7046c4b125b216f6bf1cc604cff0dc
# via
# -r requirements_formatting.txt.in
# pygithub
+1 -1
View File
@@ -2,7 +2,7 @@ black~=25.1
darker==2.1.1
PyGithub==2.6.1
cryptography>=43.0.1
urllib3>=2.6.0
urllib3>=2.5.0
requests>=2.32.4
idna>=3.7
certifi>=2024.7.4
+1 -1
+1 -1
+1 -1
+40 -13
View File
@@ -1,16 +1,16 @@
cmake_minimum_required(VERSION 3.14)
set(PROJECT_NAME FEXCore)
set (PROJECT_NAME FEXCore)
project(${PROJECT_NAME}
VERSION 0.01
LANGUAGES CXX)
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
set(ARCHITECTURE_x86_64 1)
set(_M_X86_64 1)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
set(ARCHITECTURE_arm64 1)
set(_M_ARM_64 1)
endif()
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
@@ -25,15 +25,44 @@ include(CheckIncludeFileCXX)
include(CheckCXXSourceCompiles)
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
# Useful to have for freestanding libFEXCore
add_subdirectory(External/vixl/)
include_directories(External/vixl/src/)
# Useful to have for freestanding libFEXCore
add_subdirectory(External/vixl/)
include_directories(External/vixl/src/)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
configure_file(${CMAKE_CURRENT_SOURCE_DIR}/include/git_version.h.in
set(GIT_SHORT_HASH "Unknown")
set(GIT_DESCRIBE_STRING "FEX-Unknown")
if (OVERRIDE_VERSION STREQUAL "detect")
# Find our git hash
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} rev-parse --short=7 HEAD
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_SHORT_HASH
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
endif()
else()
set(GIT_SHORT_HASH "${OVERRIDE_VERSION}")
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
endif()
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/include/git_version.h.in
${CMAKE_BINARY_DIR}/generated/git_version.h)
include_directories(${CMAKE_BINARY_DIR}/generated)
@@ -45,12 +74,10 @@ add_compile_options($<$<COMPILE_LANGUAGE:CXX>:-fno-strict-aliasing> $<$<COMPILE_
add_subdirectory(Source/)
if (NOT BUILD_STEAM_SUPPORT)
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
DESTINATION include
COMPONENT Development)
endif()
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
DESTINATION include
COMPONENT Development)
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
+165 -15
View File
@@ -118,6 +118,41 @@ def print_man_env_option(name, desc, default, no_json_key):
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
def print_man_options(options):
output_man.write(".Sh OPTIONS\n")
output_man.write(".Bl -tag -width -indent\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
default = op_vals["Default"]
value_type = op_vals["Type"]
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = op_vals["TextDefault"]
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
# Wrap the string argument in quotes
default = "'" + default + "'"
print_man_option(
short,
long,
op_vals["Desc"],
default
)
if (value_type == "strenum"):
Enums = op_vals["Enums"]
output_man.write("\\fBAvailable Options:\\fR\n")
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
output_man.write("\n.sp\n")
output_man.write(".El\n")
def print_man_environment(options):
output_man.write(".Sh ENVIRONMENT\n")
output_man.write(".Bl -tag -width -indent\n")
@@ -159,7 +194,7 @@ def print_man_environment_tail():
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
"This will override the full path",
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -173,7 +208,7 @@ def print_man_environment_tail():
"One must be careful with this option as it will override any applications that load with execve as well"
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -192,34 +227,33 @@ def print_man_environment_tail():
"PORTABLE",
[
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
"For FEX on Linux:",
"These files are instead read from <FEXPath>/fex-emu/ by default.",
"For FEXInterpreter on Linux:",
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
"For Arm64ec/Wow64 WINE builds:",
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
],
"''", True)
print_man_env_option(
"APP_CACHE_LOCATION",
[
"Allows the user to override where FEX stores and loads cache files",
"By default FEX will look in $XDG_CACHE_HOME/fex-emu/ or $HOME/.cache/fex-emu/",
"This will override the full path, trailing forward-slash is expected to exist",
],
"''", True)
def print_man_header():
header ='''.Dd {0}
.Dt FEX
.Os Linux
.Sh NAME
.Nm FEX
.Nm FEXLoader
.Nm FEXInterpreter
.Nm FEXBash
.Nd Fast x86-64 and x86 emulation.
.Sh SYNOPSIS
.Nm
.Ar <args> ...
.Op options
.Op Ar --
.Ar Application
<args> ...
.Pp
.Nm FEXInterpreter
.Ar Application
<args> ...
.Pp
.Nm FEXBash
.Ar <args> ...
@@ -327,6 +361,82 @@ def print_config_option(type, group_name, json_name, default_value, short, choic
output_argloader.write("\n");
def print_argloader_options(options):
output_argloader.write("#ifdef BEFORE_PARSE\n")
output_argloader.write("#undef BEFORE_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
default = op_vals["Default"]
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
# Wrap the string argument in quotes
default = "\"" + default + "\""
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = "\"" + op_vals["TextDefault"] + "\""
short = None
choices = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if ("Choices" in op_vals):
choices = op_vals["Choices"]
print_config_option(
op_vals["Type"],
op_group,
op_key,
default,
short,
choices,
op_vals["Desc"])
output_argloader.write("\n")
output_argloader.write("#endif\n")
def print_parse_argloader_options(options):
output_argloader.write("#ifdef AFTER_PARSE\n")
output_argloader.write("#undef AFTER_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
output_argloader.write("if (Options.is_set_by_user(\"{0}\")) {{\n".format(op_key))
value_type = op_vals["Type"]
NeedsString = False
conversion_func = "fextl::fmt::format(\"{}\", "
if ("ArgumentHandler" in op_vals):
NeedsString = True
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
if (value_type == "str"):
NeedsString = True
conversion_func = "std::move("
if (value_type == "bool"):
# boolean values need a decimal specifier. Otherwise fmt prints strings.
conversion_func = "fextl::fmt::format(\"{:d}\", "
if (value_type == "strenum"):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key))
elif (value_type == "strarray"):
# these need a bit more help
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
output_argloader.write("\t}\n")
else:
if (NeedsString):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
else:
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_envloader_options(options):
output_argloader.write("#ifdef ENVLOADER\n")
output_argloader.write("#undef ENVLOADER\n")
@@ -407,6 +517,41 @@ def print_parse_enum_options(options):
output_argloader.write("#endif\n")
def check_for_duplicate_options(options):
short_map = []
long_map = []
# Spin through all the items and see if we have a duplicate option
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
long_invert = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if (op_vals["Type"] == "bool"):
long_invert = "no-" + long
# Check for short key duplication
if (short != None):
if (short in short_map):
raise Exception("Short config '{0}' for option '{1}' has duplicate entry!".format(short, op_key))
else:
short_map.append(short)
# Check for long key duplication
if (long in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long))
else:
long_map.append(long)
# Check for long key duplication
if (long_invert != None):
if (long_invert in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long_invert))
else:
long_map.append(long_invert)
if (len(sys.argv) < 5):
sys.exit()
@@ -423,6 +568,8 @@ json_object = json.loads(json_text)
options = json_object["Options"]
unnamed_options = json_object["UnnamedOptions"]
check_for_duplicate_options(options)
# Generate config include file
output_file = open(output_filename, "w")
print_header()
@@ -434,6 +581,7 @@ output_file.close()
# Generate man file
output_man = open(output_man_page, "w")
print_man_header()
print_man_options(options)
print_man_environment(options)
print_man_tail()
@@ -441,6 +589,8 @@ output_man.close()
# Generate argument loader code
output_argloader = open(output_argumentloader_filename, "w")
print_argloader_options(options);
print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
+59 -53
View File
@@ -58,10 +58,10 @@ class OpDefinition:
JITDispatch: bool
JITDispatchOverride: str
TiedSource: int
Inline: list[str]
Arguments: list[OpArgument]
EmitValidation: list[str]
Desc: list[str]
Inline: list
Arguments: list
EmitValidation: list
Desc: list
def __init__(self):
self.Name = None
@@ -92,14 +92,19 @@ class OpDefinition:
attrs = vars(self)
print(", ".join("%s: %s" % item for item in attrs.items()))
IRTypesToCXX: dict[str, IRType] = {}
CXXTypeToIR: dict[str, IRType] = {}
IROps: list[OpDefinition] = []
IRTypesToCXX = {}
CXXTypeToIR = {}
IROps = []
IROpNameSet: set[str] = set()
IROpNameMap = {}
def is_ssa_type(op_type: str):
return op_type in {"SSA", "GPR", "GPRPair", "FPR"}
def is_ssa_type(type):
if (type == "SSA" or
type == "GPR" or
type == "GPRPair" or
type == "FPR"):
return True
return False
def parse_irtypes(irtypes):
for op_key, op_val in irtypes.items():
@@ -214,8 +219,11 @@ def parse_ops(ops):
OpArg.DefaultInitializer = DefaultInit[1][:-1]
# If SSA type then we can generate validation for this op
if OpArg.IsSSA and OpArg.Type in {"GPR", "GPRPair", "FPR"}:
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == RegClass::Invalid || WalkFindRegClass({ArgName}) == RegClass::{OpArg.Type}")
if (OpArg.IsSSA and
(OpArg.Type == "GPR" or
OpArg.Type == "GPRPair" or
OpArg.Type == "FPR")):
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
OpArg.Name = ArgName
OpArg.NameWithPrefix = NameWithPrefix
@@ -288,28 +296,21 @@ def parse_ops(ops):
#OpDef.print()
# Error on duplicate op
if OpDef.Name in IROpNameSet:
if OpDef.Name in IROpNameMap:
ExitError("Duplicate Op defined! {}".format(OpDef.Name))
IROps.append(OpDef)
IROpNameSet.add(OpDef.Name)
IROpNameMap[OpDef.Name] = 1
# Print out enum values
def print_enums(enums):
def print_enums():
output_file.write("#ifdef IROP_ENUM\n")
output_file.write("enum IROps : uint16_t {\n")
for op in IROps:
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
output_file.write("};\n")
for name, members in enums.items():
output_file.write(f"enum {name} {{\n")
for member in members:
if member:
output_file.write(f"\t{member}\n")
else:
output_file.write("\n")
output_file.write("};\n\n")
output_file.write("};\n")
output_file.write("#undef IROP_ENUM\n")
output_file.write("#endif\n\n")
@@ -407,7 +408,7 @@ def print_ir_sizes():
[[nodiscard, gnu::const]] std::string_view const& GetName(IROps Op);
[[nodiscard, gnu::const]] uint8_t GetArgs(IROps Op);
[[nodiscard, gnu::const]] uint8_t GetRAArgs(IROps Op);
[[nodiscard, gnu::const]] FEXCore::IR::RegClass GetRegClass(IROps Op);
[[nodiscard, gnu::const]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
[[nodiscard, gnu::const]] bool HasSideEffects(IROps Op);
[[nodiscard, gnu::const]] bool ImplicitFlagClobber(IROps Op);
[[nodiscard, gnu::const]] bool GetHasDest(IROps Op);
@@ -421,29 +422,30 @@ def print_ir_sizes():
def print_ir_reg_classes():
output_file.write("#ifdef IROP_REG_CLASSES_IMPL\n")
output_file.write("constexpr std::array<FEXCore::IR::RegClass, IROps::OP_LAST + 1> IRRegClasses = {\n")
output_file.write("constexpr std::array<FEXCore::IR::RegisterClassType, IROps::OP_LAST + 1> IRRegClasses = {\n")
for op in IROps:
if op.Name == "Last":
output_file.write("\tRegClass::Invalid,\n")
output_file.write("\tFEXCore::IR::InvalidClass,\n")
else:
if op.HasDest and op.DestType is None:
Class = "Invalid"
if op.HasDest and op.DestType == None:
ExitError("IR op {} has destination with no destination class".format(op.Name))
if op.HasDest and op.DestType == "SSA": # Special case SSA type
output_file.write("\tRegClass::Complex,\n")
output_file.write("\tFEXCore::IR::ComplexClass,\n")
elif op.HasDest:
output_file.write("\tRegClass::{},\n".format(op.DestType))
output_file.write("\tFEXCore::IR::{}Class,\n".format(op.DestType))
else:
# No destination so it has an invalid destination class
output_file.write("\tRegClass::Invalid, // No destination\n")
output_file.write("\tFEXCore::IR::InvalidClass, // No destination\n")
output_file.write("};\n\n")
output_file.write("// Make sure our array maps directly to the IROps enum\n")
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == RegClass::Invalid);\n\n")
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == FEXCore::IR::InvalidClass);\n\n")
output_file.write("FEXCore::IR::RegClass GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
output_file.write("FEXCore::IR::RegisterClassType GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
output_file.write("#undef IROP_REG_CLASSES_IMPL\n")
output_file.write("#endif\n\n")
@@ -566,7 +568,9 @@ def print_ir_arg_printer():
SSAArgNum = 0
FirstArg = True
for arg in op.Arguments:
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
# No point printing temporaries that we can't recover
if arg.Temporary:
continue
@@ -667,7 +671,7 @@ def print_ir_allocator_helpers():
output_file.write("\t\treturn HeaderOp->Op;\n")
output_file.write("\t}\n\n")
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n")
output_file.write("\tFEXCore::IR::RegisterClassType GetOpRegClass(const OrderedNode *Op) const {\n")
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
output_file.write("\t}\n\n")
@@ -681,21 +685,22 @@ def print_ir_allocator_helpers():
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
# Output SSA args first
for i, arg in enumerate(op.Arguments):
LastArg = i == len(op.Arguments) - 1
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
if arg.Temporary:
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
elif arg.IsSSA:
# SSA value
output_file.write("OrderedNodeWrapper {}".format(arg.Name))
else:
# User defined op that is stored
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
if arg.DefaultInitializer:
if arg.DefaultInitializer != None:
output_file.write(" = {}".format(arg.DefaultInitializer))
if not LastArg:
@@ -753,19 +758,20 @@ def print_ir_allocator_helpers():
if op.SSAArgNum:
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
for i, arg in enumerate(op.Arguments):
LastArg = i == len(op.Arguments) - 1
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
if arg.Temporary:
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
elif arg.IsSSA:
output_file.write("OrderedNode *{}".format(arg.Name))
else:
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
if arg.DefaultInitializer:
if arg.DefaultInitializer != None:
output_file.write(" = {}".format(arg.DefaultInitializer))
if not LastArg:
@@ -806,15 +812,16 @@ def print_ir_allocator_helpers():
print_validation(op)
output_file.write(f"\t\treturn _{op.Name}(")
for i, arg in enumerate(op.Arguments):
LastArg = i == len(op.Arguments) - 1
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
output_file.write(arg.Name)
if arg.IsSSA:
output_file.write("->Wrapped(ListDataBegin)")
if not LastArg:
output_file.write(", ")
output_file.write(");\n")
output_file.write("\t}\n\n")
output_file.write(");\n");
output_file.write("\t}\n\n");
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
output_file.write("#endif\n")
@@ -845,8 +852,8 @@ def print_ir_dispatcher_dispatch():
output_dispatch_file.write("#endif\n")
if len(sys.argv) < 4:
ExitError("Insufficient parameters passed to script")
if (len(sys.argv) < 4):
ExitError()
output_filename = sys.argv[2]
output_dispatcher_filename = sys.argv[3]
@@ -858,7 +865,6 @@ json_file.close()
json_object = json.loads(json_text)
json_object = {k.upper(): v for k, v in json_object.items()}
enums = json_object["ENUMS"]
ops = json_object["OPS"]
irtypes = json_object["IRTYPES"]
defines = json_object["DEFINES"]
@@ -868,7 +874,7 @@ parse_ops(ops)
output_file = open(output_filename, "w")
print_enums(enums)
print_enums()
print_ir_structs(defines)
print_ir_sizes()
print_ir_reg_classes()
+72 -55
View File
@@ -1,23 +1,23 @@
set(MAN_DIR share/man CACHE PATH "MAN_DIR")
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
set(FEXCORE_BASE_SRCS
set (FEXCORE_BASE_SRCS
Interface/Config/Config.cpp
Utils/Allocator.cpp
Utils/FileLoading.cpp
Utils/ForcedAssert.cpp
Utils/LogManager.cpp
Utils/SpinWaitLock.cpp)
Utils/SpinWaitLock.cpp
)
if (NOT MINGW)
if (NOT MINGW_BUILD)
list(APPEND FEXCORE_BASE_SRCS
Utils/Allocator/64BitAllocator.cpp)
endif()
set(SRCS
set (SRCS
Common/JitSymbols.cpp
Interface/Context/Context.cpp
Interface/Core/LookupCache.cpp
Interface/Core/CodeCache.cpp
Interface/Core/Core.cpp
Interface/Core/CPUBackend.cpp
Interface/Core/Addressing.cpp
@@ -30,6 +30,7 @@ set(SRCS
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher/X87F64.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/X86HelperGen.cpp
Interface/Core/ArchHelpers/Arm64Emitter.cpp
Interface/Core/Dispatcher/Dispatcher.cpp
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
@@ -56,6 +57,7 @@ set(SRCS
Interface/Core/X86Tables/VEXTables.cpp
Interface/Core/X86Tables/X87Tables.cpp
Interface/GDBJIT/GDBJIT.cpp
Interface/IR/AOTIR.cpp
Interface/IR/IRDumper.cpp
Interface/IR/IREmitter.cpp
Interface/IR/PassManager.cpp
@@ -64,12 +66,12 @@ set(SRCS
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/x87StackOptimizationPass.cpp
Utils/LongJump.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/Profiler.cpp)
Utils/Profiler.cpp
)
if (ARCHITECTURE_arm64)
if (_M_ARM_64)
list(APPEND SRCS Utils/ArchHelpers/Arm64.cpp)
else()
list(APPEND SRCS Utils/ArchHelpers/Arm64_stubs.cpp)
@@ -82,41 +84,42 @@ endif()
set(DEFINES -DJIT_ARM64)
if (ARCHITECTURE_x86_64)
list(APPEND DEFINES -DARCHITECTURE_x86_64=1)
if (_M_X86_64)
list(APPEND DEFINES -D_M_X86_64=1)
endif()
if (ARCHITECTURE_arm64)
list(APPEND DEFINES -DARCHITECTURE_arm64=1)
if (_M_ARM_64)
list(APPEND DEFINES -D_M_ARM_64=1)
endif()
if (ENABLE_VIXL_DISASSEMBLER)
list(APPEND DEFINES -DVIXL_DISASSEMBLER=1)
endif()
if (ARCHITECTURE_arm64 AND HAS_CLANG_PRESERVE_ALL)
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=__attribute__((preserve_all));-DFEXCORE_HAS_PRESERVE_ALL_ATTR=1")
else()
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=;-DFEXCORE_HAS_PRESERVE_ALL_ATTR=0")
endif()
set(LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter cephes_128bit)
set (LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter cephes_128bit)
if (ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
list(APPEND LIBS vixl)
list (APPEND LIBS vixl)
endif()
if (NOT MINGW)
list(APPEND LIBS dl)
if (NOT MINGW_BUILD)
list (APPEND LIBS dl)
else()
list(APPEND LIBS synchronization)
if (ARCHITECTURE_arm64ec)
list(APPEND LIBS mincore)
list (APPEND LIBS synchronization)
if (_M_ARM_64EC)
list (APPEND LIBS mincore)
endif()
endif()
# Generate config
configure_file(${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
${CMAKE_BINARY_DIR}/generated/Config/Config.json)
# Generate IR include file
@@ -131,10 +134,11 @@ add_custom_command(
OUTPUT "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
DEPENDS "${INPUT_NAME}"
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
"${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}")
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
)
set_source_files_properties(${OUTPUT_NAME} PROPERTIES GENERATED TRUE)
set_source_files_properties(${OUTPUT_NAME} PROPERTIES
GENERATED TRUE)
# Generate IR documentation
set(OUTPUT_IR_DOC "${CMAKE_BINARY_DIR}/IR.md")
@@ -143,10 +147,11 @@ add_custom_command(
OUTPUT "${OUTPUT_IR_DOC}"
DEPENDS "${INPUT_NAME}"
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
"${INPUT_NAME}" "${OUTPUT_IR_DOC}")
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py" "${INPUT_NAME}" "${OUTPUT_IR_DOC}"
)
set_source_files_properties(${OUTPUT_IR_NAME} PROPERTIES GENERATED TRUE)
set_source_files_properties(${OUTPUT_IR_NAME} PROPERTIES
GENERATED TRUE)
# Create the target
add_custom_target(IR_INC
@@ -170,12 +175,14 @@ add_custom_command(
DEPENDS "${INPUT_CONFIG_NAME}"
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py" "${INPUT_CONFIG_NAME}" "${OUTPUT_CONFIG_NAME}" "${OUTPUT_MAN_NAME}"
"${OUTPUT_CONFIG_OPTION_NAME}")
"${OUTPUT_CONFIG_OPTION_NAME}"
)
add_custom_command(
OUTPUT "${OUTPUT_MAN_NAME_COMPRESS}"
DEPENDS "${OUTPUT_MAN_NAME}"
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}")
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}"
)
set_source_files_properties(${OUTPUT_CONFIG_NAME} PROPERTIES
GENERATED TRUE)
@@ -194,10 +201,8 @@ add_custom_target(CONFIG_INC
DEPENDS "${OUTPUT_MAN_NAME}"
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
if (NOT BUILD_STEAM_SUPPORT)
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
endif()
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
# Add in diagnostic colours if the option is available.
# Ninja code generator will kill colours if this isn't here
@@ -219,7 +224,8 @@ function(AddDefaultOptionsToTarget Name)
target_compile_definitions(${Name} PRIVATE ${DEFINES})
add_dependencies(${Name} CONFIG_INC IR_INC)
target_compile_options(${Name} PRIVATE
target_compile_options(${Name}
PRIVATE
-Wall
-Werror=cast-qual
-Werror=ignored-qualifiers
@@ -227,20 +233,31 @@ function(AddDefaultOptionsToTarget Name)
-Wno-trigraphs
-ffunction-sections
-fwrapv)
-fwrapv
)
if (GCC_COLOR)
target_compile_options(${Name} PRIVATE "-fdiagnostics-color=always")
target_compile_options(${Name}
PRIVATE
"-fdiagnostics-color=always")
endif()
if (CLANG_COLOR)
target_compile_options(${Name} PRIVATE "-fcolor-diagnostics")
target_compile_options(${Name}
PRIVATE
"-fcolor-diagnostics")
endif()
LinkerGC(${Name})
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
target_link_options(${Name}
PRIVATE
"LINKER:--gc-sections"
"LINKER:--strip-all"
"LINKER:--as-needed"
)
endif()
endfunction()
# Build FEXCore_Base static library
# Build FEXCore_Config static library
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
target_link_libraries(FEXCore_Base ${LIBS})
AddDefaultOptionsToTarget(FEXCore_Base)
@@ -249,34 +266,34 @@ if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
target_link_libraries(FEXCore_Base TracyClient)
endif()
function(AddObject Name)
add_library(${Name} OBJECT ${SRCS})
function(AddObject Name Type)
add_library(${Name} ${Type} ${SRCS})
target_link_libraries(${Name} PRIVATE FEXCore_Base)
target_link_libraries(${Name} FEXCore_Base)
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
AddDefaultOptionsToTarget(${Name})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
endfunction()
function(AddLibrary Name Type)
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
target_link_libraries(${Name} FEXCore_Base)
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
# During generation of the import library (dll.a), MinGW needs some extra symbols from libraries
# such as fmt, which are propagated by FEXCore_Base. Wonderful.
if (MINGW)
target_link_libraries(${Name} FEXCore_Base)
endif()
AddDefaultOptionsToTarget(${Name})
endfunction()
AddObject(${PROJECT_NAME}_object)
AddObject(${PROJECT_NAME}_object OBJECT)
AddLibrary(${PROJECT_NAME} STATIC)
AddLibrary(${PROJECT_NAME}_shared SHARED)
if (NOT MINGW AND NOT BUILD_STEAM_SUPPORT)
install(TARGETS ${PROJECT_NAME}_shared LIBRARY
DESTINATION ${CMAKE_INSTALL_LIBDIR}
COMPONENT Libraries)
if (NOT MINGW_BUILD)
install(TARGETS ${PROJECT_NAME}_shared
LIBRARY
DESTINATION ${CMAKE_INSTALL_LIBDIR}
COMPONENT Libraries)
endif()
# Meta-library to link jemalloc libraries enabled in the build configuration.
@@ -291,7 +308,7 @@ if (ENABLE_JEMALLOC_GLIBC_ALLOC)
target_link_libraries(JemallocLibs INTERFACE FEX_jemalloc_glibc)
endif()
if (NOT MINGW)
if (NOT MINGW_BUILD)
# Dummy project to use for host tools.
# This overrides use of jemalloc in FEXCore with the normal glibc allocator.
add_library(JemallocDummy STATIC Utils/AllocatorHooks.cpp)
+2 -3
View File
@@ -1,6 +1,5 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/memory.h>
@@ -13,7 +12,7 @@ namespace FEXCore {
// Buffered JIT symbol tracking.
struct JITSymbolBuffer {
// Maximum buffer size to ensure we are a page in size.
constexpr static size_t BUFFER_SIZE = FEXCore::Utils::FEX_PAGE_SIZE - (8 * 2);
constexpr static size_t BUFFER_SIZE = 4096 - (8 * 2);
// Maximum distance until the end of the buffer to do a write.
constexpr static size_t NEEDS_WRITE_DISTANCE = BUFFER_SIZE - 64;
// Maximum time threshhold to wait before a buffer write occurs.
@@ -28,7 +27,7 @@ struct JITSymbolBuffer {
size_t Offset {};
char Buffer[BUFFER_SIZE] {};
};
static_assert(sizeof(JITSymbolBuffer) == FEXCore::Utils::FEX_PAGE_SIZE, "Ensure this is one page in size");
static_assert(sizeof(JITSymbolBuffer) == 4096, "Ensure this is one page in size");
class JITSymbols final {
public:
+10 -10
View File
@@ -4,9 +4,9 @@
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/BitUtils.h>
#include "cephes_128bit.h"
#include <bit>
#include <cmath>
#include <cstring>
#include <stdint.h>
@@ -19,7 +19,7 @@ extern "C" {
}
struct FEX_PACKED X80SoftFloat {
#ifdef ARCHITECTURE_x86_64
#ifdef _M_X86_64
// Define this to push some operations to x87
// Only useful to see if precision loss is killing something
// #define DEBUG_X86_FLOAT
@@ -30,7 +30,7 @@ struct FEX_PACKED X80SoftFloat {
#define BIGFLOAT float128_t
#define BIGFLOATSIZE 16
#endif
#elif defined(ARCHITECTURE_arm64)
#elif defined(_M_ARM_64)
#define BIGFLOAT float128_t
#define BIGFLOATSIZE 16
#else
@@ -501,12 +501,12 @@ struct FEX_PACKED X80SoftFloat {
float ToF32(softfloat_state* state) const {
const float32_t Result = extF80_to_f32(state, *this);
return std::bit_cast<float>(Result);
return FEXCore::BitCast<float>(Result);
}
double ToF64(softfloat_state* state) const {
const float64_t Result = extF80_to_f64(state, *this);
return std::bit_cast<double>(Result);
return FEXCore::BitCast<double>(Result);
}
FEXCore::VectorRegType ToVector() const {
@@ -518,7 +518,7 @@ struct FEX_PACKED X80SoftFloat {
BIGFLOAT ToFMax(softfloat_state* state) const {
#if BIGFLOATSIZE == 16
const float128_t Result = extF80_to_f128(state, *this);
return std::bit_cast<BIGFLOAT>(Result);
return FEXCore::BitCast<BIGFLOAT>(Result);
#else
BIGFLOAT result {};
memcpy(&result, this, sizeof(result));
@@ -577,18 +577,18 @@ struct FEX_PACKED X80SoftFloat {
}
X80SoftFloat(softfloat_state* state, const float rhs) {
*this = f32_to_extF80(state, std::bit_cast<float32_t>(rhs));
*this = f32_to_extF80(state, FEXCore::BitCast<float32_t>(rhs));
}
X80SoftFloat(softfloat_state* state, const double rhs) {
*this = f64_to_extF80(state, std::bit_cast<float64_t>(rhs));
*this = f64_to_extF80(state, FEXCore::BitCast<float64_t>(rhs));
}
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
#if BIGFLOATSIZE == 16
*this = f128_to_extF80(state, std::bit_cast<float128_t>(rhs));
*this = f128_to_extF80(state, FEXCore::BitCast<float128_t>(rhs));
#else
*this = std::bit_cast<long double>(rhs);
*this = FEXCore::BitCast<long double>(rhs);
#endif
}
+47 -11
View File
@@ -2,23 +2,59 @@
#pragma once
#include <FEXCore/fextl/string.h>
#include <concepts>
#include <cstdint>
#include <string_view>
#include <optional>
namespace FEXCore::StrConv {
template<std::integral T>
bool Conv(std::string_view Value, T* Result) {
if constexpr (std::is_signed_v<T>) {
*Result = static_cast<T>(std::strtoll(Value.data(), nullptr, 0));
} else {
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
}
inline bool Conv(std::string_view Value, bool* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
template<typename T, typename = std::enable_if_t<std::is_enum_v<T>, T>>
bool Conv(std::string_view Value, T* Result) {
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
inline bool Conv(std::string_view Value, uint8_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int8_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint16_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int16_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint32_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int32_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint64_t* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int64_t* Result) {
*Result = std::strtoll(Value.data(), nullptr, 0);
return true;
}
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
inline bool Conv(std::string_view Value, T* Result) {
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
return true;
}
+3 -5
View File
@@ -1,11 +1,9 @@
// SPDX-License-Identifier: MIT
#pragma once
#ifdef ARCHITECTURE_x86_64
#ifdef _M_X86_64
#include <xmmintrin.h>
#include <immintrin.h>
#else
#include <cstdint>
#endif
namespace FEXCore {
@@ -13,7 +11,7 @@ struct VectorScalarF64Pair {
double val[2];
};
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
// Can't use uint8x16_t directly from arm_neon.h here.
// Overrides softfloat-3e's defines which causes problems.
using VectorRegType = __attribute__((neon_vector_type(16))) uint8_t;
@@ -25,7 +23,7 @@ static inline VectorRegPairType MakeVectorRegPair(VectorRegType low, VectorRegTy
return VectorRegPairType {low, high};
}
#elif defined(ARCHITECTURE_x86_64)
#elif defined(_M_X86_64)
using VectorRegType = __m128i;
using VectorRegPairType = __m256i;
+14 -19
View File
@@ -30,14 +30,14 @@ class Context;
}
namespace FEXCore::Config {
namespace detail {
namespace DefaultValues {
#define P(x) x
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
#include <FEXCore/Config/ConfigValues.inl>
} // namespace detail
} // namespace DefaultValues
enum Paths {
PATH_DATA_DIR_LOCAL = 0,
@@ -134,7 +134,7 @@ public:
void Load();
template<typename T>
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<StringArrayType, T>)
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<DefaultValues::Type::StringArrayType, T>)
std::optional<T> GetConv(ConfigOption Option) {
const auto it = OptionMap.find(Option);
if (it == OptionMap.end()) {
@@ -142,7 +142,7 @@ public:
}
const auto& Value = it->second;
LOGMAN_THROW_A_FMT(!std::holds_alternative<StringArrayType>(Value), "Tried to get config of invalid type!");
LOGMAN_THROW_A_FMT(!std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
if (std::holds_alternative<T>(Value)) [[likely]] {
return std::get<T>(Value);
@@ -165,7 +165,7 @@ public:
private:
void MergeConfigMap(const LayerOptions& Options);
void MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value);
void MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value);
};
void MetaLayer::Load() {
@@ -181,7 +181,7 @@ void MetaLayer::Load() {
}
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value) {
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value) {
// Environment variables need a bit of additional work
// We want to merge the arrays rather than overwrite entirely
auto MetaEnvironment = OptionMap.find(Option);
@@ -193,7 +193,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Stri
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
const auto AddToMap = [&LookupMap](const StringArrayType& Value) {
const auto AddToMap = [&LookupMap](const DefaultValues::Type::StringArrayType& Value) {
for (const auto& EnvVar : Value) {
const auto ItEq = EnvVar.find_first_of('=');
if (ItEq == fextl::string::npos) {
@@ -209,7 +209,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Stri
}
};
AddToMap(std::get<StringArrayType>(MetaEnvironment->second));
AddToMap(std::get<DefaultValues::Type::StringArrayType>(MetaEnvironment->second));
AddToMap(Value);
// Now with the two layers merged in the map
@@ -225,8 +225,8 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
// Insert this layer's options, overlaying previous options that exist here
for (auto& it : Options) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
LOGMAN_THROW_A_FMT(std::holds_alternative<StringArrayType>(it.second), "Tried to get config of invalid type!");
MergeEnvironmentVariables(it.first, std::get<StringArrayType>(it.second));
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(it.second), "Tried to get config of invalid type!");
MergeEnvironmentVariables(it.first, std::get<DefaultValues::Type::StringArrayType>(it.second));
} else {
OptionMap.insert_or_assign(it.first, it.second);
}
@@ -423,7 +423,7 @@ bool Exists(ConfigOption Option) {
return Meta->OptionExists(Option);
}
std::optional<StringArrayType*> All(ConfigOption Option) {
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
return Meta->All(Option);
}
@@ -436,12 +436,6 @@ std::optional<T> GetConv(ConfigOption Option) {
return Meta->GetConv<T>(Option);
}
template std::optional<bool> GetConv(ConfigOption Option);
template std::optional<uint8_t> GetConv(ConfigOption Option);
template std::optional<int32_t> GetConv(ConfigOption Option);
template std::optional<uint32_t> GetConv(ConfigOption Option);
template std::optional<uint64_t> GetConv(ConfigOption Option);
void Set(ConfigOption Option, std::string_view Data) {
Meta->Set(Option, Data);
}
@@ -497,12 +491,13 @@ template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t De
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
template<typename T>
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List) {
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List) {
auto Value = FEXCore::Config::All(Option);
List->clear();
if (Value) {
*List = **Value;
}
}
template void Value<StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
template void Value<DefaultValues::Type::StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option,
DefaultValues::Type::StringArrayType* List);
} // namespace FEXCore::Config
+42 -75
View File
@@ -4,6 +4,7 @@
"Multiblock": {
"Type": "bool",
"Default": "true",
"ShortArg": "m",
"Desc": [
"Controls multiblock code compilation",
"Can cause long JIT compilation times and stutter"
@@ -12,24 +13,11 @@
"MaxInst": {
"Type": "int32",
"Default": "5000",
"ShortArg": "n",
"Desc": [
"Maximum number of instruction to store in a block"
]
},
"EnableCodeCachingWIP": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enable the code caching subsystem"
]
},
"EnableCodeCacheValidation": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enable expensive validation when loading code caches"
]
},
"HostFeatures": {
"Type": "strenum",
"Default": "FEXCore::Config::HostFeatures::OFF",
@@ -73,9 +61,7 @@
"ENABLEWFXT": "enablewfxt",
"DISABLEWFXT": "disablewfxt",
"ENABLE3DNOW": "enable3dnow",
"DISABLE3DNOW": "disable3dnow",
"ENABLESSE4A": "enablesse4a",
"DISABLESSE4A": "disablesse4a"
"DISABLE3DNOW": "disable3dnow"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
@@ -98,8 +84,7 @@
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it"
"\t{enable,disable}3dnow: Will force enable or disable 3DNow even if the host doesn't support it"
]
},
"SmallTSCScale": {
@@ -108,19 +93,13 @@
"Desc": [
"Scales the cycle counter on systems that have low frequencies."
]
},
"CPUFeatureRegisters": {
"Type": "str",
"Default": "",
"Desc": [
"Allows overriding cpu feature flags for manual testing"
]
}
},
"Emulation": {
"RootFS": {
"Type": "str",
"Default": "",
"ShortArg": "R",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
@@ -135,6 +114,7 @@
"ThunkHostLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
]
@@ -142,6 +122,7 @@
"ThunkGuestLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
@@ -149,6 +130,7 @@
"ThunkConfig": {
"Type": "str",
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks.",
"This can be a filesystem path",
@@ -163,6 +145,7 @@
"Env": {
"Type": "strarray",
"Default": "",
"ShortArg": "E",
"Desc": [
"Adds an environment variable to the emulated environment."
]
@@ -170,6 +153,7 @@
"HostEnv": {
"Type": "strarray",
"Default": "",
"ShortArg": "H",
"Desc": [
"Adds an environment variable to the host environment.",
"This can be useful for setting environment variables that thunks can pick up.",
@@ -182,50 +166,13 @@
"Desc": [
"Allows the user to pass additional arguments to the application"
]
},
"DisableL2Cache": {
"Type": "bool",
"Default": "false",
"Desc": [
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
"Can potentially introduce more stutters."
]
},
"DynamicL1Cache": {
"Type": "bool",
"Default": "false",
"Desc": [
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
"Can potentially introduce more stutters."
]
},
"DynamicL1CacheIncreaseCountHeuristic": {
"Type": "uint64",
"Default": "250",
"Desc": [
"Threshold of lookups per second that the L1 dynamic cache should increase its size.",
"Lower numbers means more aggressive scaling upward to the maximum size.",
"Higher numbers means more conservative scaling, using less memory.",
"Can potentially introduce stutters, more likely the higher the number.",
"Don't have this number smaller than the decrease count!"
]
},
"DynamicL1CacheDecreaseCountHeuristic": {
"Type": "uint64",
"Default": "50",
"Desc": [
"Threshold of lookups per second that the L1 dynamic cache should decrease its size.",
"The higher the number, the more aggressively it reduces the L1 cache size.",
"Lower numbers means more conservative memory savings.",
"Can potentially introduce more stutters, more likely the higher the number.",
"Don't have this number larger than the increase count!"
]
}
},
"Debug": {
"SingleStep": {
"Type": "bool",
"Default": "false",
"ShortArg": "S",
"Desc": [
"Single stepping configuration."
]
@@ -233,6 +180,7 @@
"GdbServer": {
"Type": "bool",
"Default": "false",
"ShortArg": "G",
"Desc": [
"Enables the GDB server."
]
@@ -266,6 +214,7 @@
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"ShortArg": "g",
"Desc": [
"When the test harness ends, print the GPR state."
]
@@ -273,6 +222,7 @@
"O0": {
"Type": "bool",
"Default": "false",
"ShortArg": "O0",
"Desc": [
"Disables optimizations passes for debugging."
]
@@ -362,6 +312,7 @@
"SilentLog": {
"Type": "bool",
"Default": "true",
"ShortArg": "s",
"Desc": [
"Disables logging"
]
@@ -369,9 +320,10 @@
"OutputLog": {
"Type": "str",
"Default": "server",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
"[stderr, server, <Filename>]"
"[stdout, stderr, server, <Filename>]"
]
},
"TelemetryDirectory": {
@@ -389,13 +341,6 @@
"Enables FEX's low-overhead sampling profile statistics.",
"Requires a supported version of Mangohud to see the results"
]
},
"EnableGpuvisProfiling": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enables profiling when FEX was built with the gpuvis profiler backend."
]
}
},
"Hacks": {
@@ -450,11 +395,12 @@
"This is required to ensure a split-lock doesn't tear inside the process"
]
},
"KernelUnalignedAtomicBackpatching": {
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
"Desc": [
"When the kernel unaligned atomic handler is enabled, use backpatching to reduce kernel context switches."
"Automatically enables TSO when shared memory is used.",
"Should work without issues in most cases."
]
},
"VolatileMetadata": {
@@ -462,7 +408,7 @@
"Default": "true",
"Desc": [
"Use volatile metadata in PE files to inform TSO instructions when available.",
"When metadata is unavailable falls back to the currently enabled TSO options."
"When metadata is unavailable falls back to the currently enabled TSO options."
]
},
"X87ReducedPrecision": {
@@ -472,6 +418,23 @@
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
]
},
"ABILocalFlags": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around flags.",
"Assumes flags are not used across cals.",
"Hand-written assembly can violate this assumption."
]
},
"ParanoidTSO": {
"Type": "bool",
"Default": "false",
"Desc": [
"Makes TSO operations even more strict.",
"Forces vector loadstores to also become atomic."
]
},
"StallProcess": {
"Type": "bool",
"Default": "false",
@@ -551,6 +514,10 @@
},
"UnnamedOptions": {
"Misc": {
"IS_INTERPRETER": {
"Type": "bool",
"Default": "false"
},
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false"
@@ -1,7 +1,6 @@
// SPDX-License-Identifier: MIT
#include "Interface/Context/Context.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Core/CoreState.h>
+60 -98
View File
@@ -4,47 +4,54 @@
#include "Common/JitSymbols.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/AOTIR.h"
#include <Interface/IR/IntrusiveIRList.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <stdint.h>
#include <atomic>
#include <cstddef>
#include <cstdint>
#include <mutex>
#include <optional>
#include <shared_mutex>
namespace FEXCore {
class SignalDelegator;
class CodeLoader;
class ThunkHandler;
struct LookupCacheWriteLockToken;
namespace Core {
struct DebugData;
struct InternalThreadState;
} // namespace Core
namespace CPU {
class Arm64JITCore;
class Dispatcher;
} // namespace CPU
namespace HLE {
class SourcecodeResolver;
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
} // namespace HLE
} // namespace FEXCore
namespace FEXCore::IR {
namespace Validation {
class IRValidation;
}
} // namespace FEXCore::IR
namespace FEXCore::Context {
struct FEX_PACKED ExitFunctionLinkData {
uint64_t HostCode;
@@ -61,66 +68,10 @@ struct CustomIRResult {
, Data(Data) {}
};
using BlockDelinkerFunc = void (*)(FEXCore::Context::ExitFunctionLinkData* Record);
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
class CodeCache : public AbstractCodeCache {
public:
CodeCache(ContextImpl&);
~CodeCache();
ContextImpl& CTX;
fextl::unique_ptr<ContextImpl> ValidationCTX;
fextl::unique_ptr<Core::InternalThreadState> ValidationThread;
FEXCore::Core::CPUState::gdt_segment ValidationGDT[32] {};
bool IsGeneratingCache = false;
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
uint64_t ComputeCodeMapId(std::string_view Filename, int FD) override;
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
bool LoadData(Core::InternalThreadState*, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
/**
* Performs expensive extra validation on the loaded code cache data.
*
* This kicks off an in-process recompile of all cached blocks and compares
* them with the cached data. Differences will be reported as fatal errors,
* which can uncover bugs like for example:
* - mismatches of the JIT configuration used during cache generation
* - hidden position dependencies due to missing FEX relocations
* - incorrect instruction padding
*/
void Validate(const ExecutableFileSectionInfo&, fextl::set<uint64_t> GuestBlocks, const fextl::set<uint64_t>& HostBlocks,
std::span<std::byte> CachedCode);
void InitiateCacheGeneration() override {
IsGeneratingCache = true;
}
/**
* Applies a set of FEX relocations to the given code section.
*
* FEX relocations describe runtime-dependencies of FEX-generated code.
* When loading a code cache, they are used to move cached code to the
* dynamically chosen base address of the guest binary.
*
* Conversely, relocations are applied in reverse when writing code caches
* to ensure consistency across generation runs.
*
* Note that FEX relocations are unrelated to ELF/PE relocations.
*
* @param GuestDelta Guest address offset to apply to RIP-relative data
* @param ForStorage True for serializing data (producing deterministic output); false for de-serializing it (resolving dynamic symbols)
*
* @return Returns true on success
*/
[[nodiscard]]
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations, bool ForStorage);
};
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
public:
// Context base class implementation.
bool InitCore() override;
@@ -190,28 +141,21 @@ public:
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
CodeCache& GetCodeCache() override {
return CodeCache;
}
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
void SetCodeMapWriter(fextl::unique_ptr<CodeMapWriter> Writer) override {
CodeMapWriter = std::move(Writer);
}
void FinalizeAOTIRCache() override {}
void FlushAndCloseCodeMap() override {
if (CodeMapWriter) {
CodeMapWriter.reset();
}
}
void OnCodeBufferAllocated(const std::shared_ptr<CPU::CodeBuffer>&) override;
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
void InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) override;
void InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start,
uint64_t Length) override;
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
return CodeInvalidationMutex;
}
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
@@ -233,6 +177,13 @@ public:
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
public:
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
friend class FEXCore::IR::Validation::IRValidation;
struct {
uint64_t VirtualMemSize {1ULL << 36};
uint64_t TSCScale = 0;
@@ -245,8 +196,10 @@ public:
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
@@ -254,6 +207,7 @@ public:
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
@@ -273,13 +227,14 @@ public:
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
FEXCore::ThunkHandler* ThunkHandler {};
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CodeCache CodeCache;
fextl::unique_ptr<CodeMapWriter> CodeMapWriter;
SignalDelegator* SignalDelegation {};
X86GeneratedCode X86CodeGen;
ContextImpl(const FEXCore::HostFeatures& Features);
static bool ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
// This is used as a replacement for the SMC writes in the mono callsite backpatcher that avoids atomic operations
@@ -314,9 +269,9 @@ public:
FEXCore::JITSymbols Symbols;
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator {"FEXMem_OpDispatcher"};
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator {"FEXMem_Frontend"};
FEXCore::Utils::PooledAllocatorVirtualWithGuard CPUBackendAllocator {"FEXMem_CPUBackend"};
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
FEXCore::Utils::PooledAllocatorVirtual CPUBackendAllocator;
// If Atomic-based TSO emulation is enabled or not.
bool IsAtomicTSOEnabled() const {
@@ -357,10 +312,17 @@ protected:
AtomicTSOEmulationEnabled = false;
VectorAtomicTSOEmulationEnabled = false;
MemcpyAtomicTSOEmulationEnabled = false;
} else if (Config.ParanoidTSO) {
AtomicTSOEmulationEnabled = true;
VectorAtomicTSOEmulationEnabled = true;
MemcpyAtomicTSOEmulationEnabled = true;
} else {
AtomicTSOEmulationEnabled = Config.TSOEnabled;
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
MemcpyAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.MemcpySetTSOEnabled;
// Atomic TSO emulation only enabled if the config option is enabled.
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
}
}
@@ -374,6 +336,9 @@ private:
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
IR::AOTIRCaptureCache IRCaptureCache;
bool IsMemoryShared = false;
bool SupportsHardwareTSO = false;
bool AtomicTSOEmulationEnabled = true;
bool VectorAtomicTSOEmulationEnabled = false;
@@ -386,8 +351,8 @@ private:
std::atomic<bool> HasCustomIRHandlers {};
struct CustomIRHandlerEntry final {
CustomIREntrypointHandler Handler;
void* Creator;
void* Data;
void *Creator;
void *Data;
};
fextl::unordered_map<uint64_t, CustomIRHandlerEntry> CustomIRHandlers;
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
@@ -395,8 +360,5 @@ private:
bool MonoDetected = false;
std::atomic<uint64_t> MonoBackpatcherBlock;
std::mutex CodeBufferListLock;
fextl::vector<std::weak_ptr<CPU::CodeBuffer>> CodeBufferList;
};
} // namespace FEXCore::Context
+11 -13
View File
@@ -7,7 +7,7 @@
namespace FEXCore::IR {
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
Ref Tmp = A.Base;
if (A.Offset) {
@@ -51,8 +51,8 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
return Tmp ?: IREmit->Constant(0);
}
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
bool Vector, IR::OpSize AccessSize) {
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
IR::OpSize AccessSize) {
const auto Is32Bit = GPRSize == OpSize::i32Bit;
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
@@ -103,7 +103,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset),
.IndexType = MemOffsetType::SXTX,
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = 1,
};
}
@@ -111,17 +111,15 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
if (AtomicTSO) {
// TODO: LRCPC3 support for vector Imm9.
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
AddressMode B = A;
// ScaledRegisterLoadstore
if (B.Index && B.Segment) {
B.Base = IREmit->Add(GPRSize, B.Base, B.Segment);
} else if (B.Segment) {
B.Index = B.Segment;
B.IndexScale = 1;
if (A.Index && A.Segment) {
A.Base = IREmit->Add(GPRSize, A.Base, A.Segment);
} else if (A.Segment) {
A.Index = A.Segment;
A.IndexScale = 1;
}
return B;
return A;
}
if (Vector || !AtomicTSO) {
@@ -136,7 +134,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset),
.IndexType = MemOffsetType::SXTX,
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = 1,
};
}
+6 -7
View File
@@ -11,18 +11,17 @@ struct AddressMode {
Ref Segment {nullptr};
Ref Base {nullptr};
Ref Index {nullptr};
int64_t Offset = 0;
MemOffsetType IndexType = MemOffsetType::SXTX;
MemOffsetType IndexType = MEM_OFFSET_SXTX;
uint8_t IndexScale = 1;
int64_t Offset = 0;
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
IR::OpSize AddrSize;
bool NonTSO;
};
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
bool Vector, IR::OpSize AccessSize);
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
IR::OpSize AccessSize);
} // namespace FEXCore::IR
}; // namespace FEXCore::IR
@@ -1,10 +1,10 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "FEXCore/Core/X86Enums.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Context/Context.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
@@ -41,7 +41,7 @@ namespace FEXCore::CPU {
// r19-r29 and SP.
namespace x64 {
#ifndef ARCHITECTURE_arm64ec
#ifndef _M_ARM_64EC
// All but x19 and x29 are caller saved
// Note that rax/rdx are rearranged here so we can coalesce cmpxchg.
constexpr std::array<ARMEmitter::Register, 18> SRA = {
@@ -417,34 +417,18 @@ FEXCore::X86State::X86Reg Arm64Emitter::GetX86RegRelationToARMReg(ARMEmitter::Re
return FEXCore::X86State::X86Reg::REG_INVALID;
}
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad, int MaxBytes) {
bool NOPPad = false;
if (Pad == PadType::DOPAD) {
NOPPad = true;
} else if (Pad == PadType::NOPAD) {
NOPPad = false;
} else if (Pad == PadType::AUTOPAD) {
// Force NOP padding to ensure relocated constants always have enough encoding space available
NOPPad = EnableCodeCaching;
}
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
const auto UpperBound = Is64Bit ? 4 : 2;
int Segments = MaxBytes ? (MaxBytes / 2) : UpperBound;
LOGMAN_THROW_A_FMT(MaxBytes >= 0 && MaxBytes <= (UpperBound * 2) && (MaxBytes & 1) == 0,
"MaxBytes must be bounded in the range of [0, {}] and 16-bit aligned", UpperBound);
// If MaxBytes specified then make sure to sanity check incoming data.
LOGMAN_THROW_A_FMT(MaxBytes == 0 || (Constant >> (MaxBytes * 8)) == 0, "MaxBytes provided but data can't fit within provided range.");
int Segments = Is64Bit ? 4 : 2;
if (Is64Bit && ((~Constant) >> 16) == 0) {
movn(s, Reg, (~Constant) & 0xFFFF);
if (NOPPad) {
nop();
nop();
nop();
}
movn(s, Reg, (~Constant) & 0xFFFF);
return;
}
@@ -452,17 +436,17 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
// If the upper 32-bits is all zero, we can now switch to a 32-bit move.
s = ARMEmitter::Size::i32Bit;
Is64Bit = false;
Segments = std::min(Segments, 2);
Segments = 2;
}
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
movn(s, Reg.W(), (~Constant) & 0xFFFF);
if (NOPPad) {
nop();
nop();
nop();
}
movn(s, Reg.W(), (~Constant) & 0xFFFF);
return;
}
@@ -483,24 +467,24 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
// `movz` is better than `orr` since hardware will rename or merge if possible when `movz` is used.
const auto IsImm = ARMEmitter::Emitter::IsImmLogical(Constant, RegSizeInBits(s));
if (IsImm) {
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
if (NOPPad) {
nop();
nop();
nop();
}
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
return;
}
}
// If we can't handle negatives with the orr, try with movn+movk
if (Is64Bit && ((~Constant) >> 32) == 0) {
movn(s, Reg, (~Constant) & 0xFFFF);
movk(s, Reg, (Constant >> 16) & 0xFFFF, 16);
if (NOPPad) {
nop();
nop();
}
movn(s, Reg, (~Constant) & 0xFFFF);
movk(s, Reg, (Constant >> 16) & 0xFFFF, 16);
return;
}
@@ -802,7 +786,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
auto TmpReg = *OptionalReg;
auto TmpReg2 = *OptionalReg2;
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
// Load STATE in from the CPU area as x28 is not callee saved in the ARM64EC ABI.
ldr(TmpReg.X(), ARMEmitter::Reg::r18, TEB_CPU_AREA_OFFSET);
ldr(STATE, TmpReg, CPU_AREA_EMULATOR_DATA_OFFSET);
@@ -1,38 +1,36 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Config/Config.h>
#include "FEXCore/Utils/EnumUtils.h"
#include "Interface/Core/JIT/Relocations.h"
#ifdef VIXL_DISASSEMBLER
#include <aarch64/disasm-aarch64.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/vector.h>
#endif
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#include <aarch64/simulator-constants-aarch64.h>
#endif
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/vector.h>
#include <CodeEmitter/Emitter.h>
#include <CodeEmitter/Registers.h>
#include <cstddef>
#include <cstdint>
#include <optional>
#include <span>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::X86State {
enum X86Reg : uint32_t;
}
namespace FEXCore::CPU {
// Contains the address to the currently available CPU state
constexpr auto STATE = ARMEmitter::XReg::x28;
#ifndef ARCHITECTURE_arm64ec
#ifndef _M_ARM_64EC
// GPR temporaries. Only x3 can be used across spill boundaries
// so if these ever need to change, be very careful about that.
constexpr auto TMP1 = ARMEmitter::XReg::x0;
@@ -106,20 +104,9 @@ constexpr ARMEmitter::PRegister PRED_TMP_32B = ARMEmitter::PReg::p7;
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
class Arm64Emitter : public ARMEmitter::Emitter {
public:
protected:
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
enum class PadType {
// Explicitly does not need padding, even if code-caching is enabled.
NOPAD,
// Explicitly needs padding, even if code-caching is disabled.
DOPAD,
// Choose to pad or not depending on if code-caching is enabled.
AUTOPAD,
};
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad = PadType::NOPAD, int MaxBytes = 0);
protected:
FEXCore::Context::ContextImpl* EmitterCTX;
std::span<const ARMEmitter::Register> StaticRegisters {};
@@ -129,6 +116,8 @@ protected:
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
uint32_t PairRegisters = 0;
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
// Correlate an ARM register back to an x86 register index.
@@ -281,8 +270,6 @@ protected:
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
#endif
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
};
} // namespace FEXCore::CPU
+16 -21
View File
@@ -1,17 +1,14 @@
// SPDX-License-Identifier: MIT
#include "FEXCore/IR/IR.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/Utils/PrctlUtils.h>
#include <cstdint>
#include "LookupCache.h"
#ifndef _WIN32
#include <linux/prctl.h>
#include <sys/prctl.h>
#endif
@@ -277,37 +274,37 @@ namespace CPU {
: ThreadState(ThreadState)
, CodeBuffers(CodeBuffers) {
auto& Ptrs = ThreadState->CurrentFrame->Pointers;
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
// Initialize named vector constants.
for (size_t i = 0; i < FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX; ++i) {
Ptrs.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
}
// Copy named vector constants.
memcpy(Ptrs.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
// Initialize Indexed named vector constants.
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] =
reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] =
reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] =
reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] =
reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] =
reinterpret_cast<uint64_t>(DPPS_MASK.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] =
reinterpret_cast<uint64_t>(DPPD_MASK.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] =
reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
#ifndef FEX_DISABLE_TELEMETRY
// Fill in telemetry values
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
auto& Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
Ptrs.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(&Telem);
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(&Telem);
}
#endif
}
@@ -360,8 +357,6 @@ namespace CPU {
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
}
FEXCore::Allocator::VirtualName("FEXMemJIT", reinterpret_cast<void*>(Ptr), Size);
LookupCache = fextl::make_unique<GuestToHostMap>();
}
@@ -400,7 +395,7 @@ namespace CPU {
Latest = Buffer;
LatestOffset = 0;
OnCodeBufferAllocated(Buffer);
OnCodeBufferAllocated(*Buffer);
return Buffer;
}
+13 -6
View File
@@ -17,10 +17,6 @@ $end_info$
#include <cstdint>
namespace FEXCore::CPU {
union Relocation;
}
namespace FEXCore {
namespace IR {
@@ -81,7 +77,7 @@ namespace CPU {
// Protects writes to the latest CodeBuffer and changes to LatestOffset
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
virtual void OnCodeBufferAllocated(CodeBuffer&) {};
private:
fextl::shared_ptr<CodeBuffer> Latest;
@@ -161,7 +157,18 @@ namespace CPU {
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) = 0;
/**
* @brief Relocates a block of code from the JIT code object cache
*
* @param Entry - RIP of the entry
* @param SerializationData - Serialization data referring to the object cache for `Entry`
*
* @return An executable function pointer relocated from the cache object
*/
[[nodiscard]]
virtual void* RelocateJITObjectCode(uint64_t /* Entry */, const CodeSerialize::CodeObjectFileSection* /* SerializationData */) {
return nullptr;
}
virtual void ClearCache() {}
+77 -97
View File
@@ -14,7 +14,6 @@ $end_info$
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Utils/FileLoading.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/Syscalls.h>
@@ -24,7 +23,7 @@ $end_info$
namespace FEXCore {
namespace ProductNames {
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
static const char ARM_UNKNOWN[] = "Unknown ARM CPU";
static const char ARM_A57[] = "Cortex-A57";
static const char ARM_A72[] = "Cortex-A72";
@@ -44,15 +43,12 @@ namespace ProductNames {
static const char ARM_A715[] = "Cortex-A715";
static const char ARM_A720[] = "Cortex-A720";
static const char ARM_A725[] = "Cortex-A725";
static const char ARM_C1Pro[] = "C1-Pro";
static const char ARM_C1Premium[] = "C1-Premium";
static const char ARM_X1[] = "Cortex-X1";
static const char ARM_X1C[] = "Cortex-X1C";
static const char ARM_X2[] = "Cortex-X2";
static const char ARM_X3[] = "Cortex-X3";
static const char ARM_X4[] = "Cortex-X4";
static const char ARM_X925[] = "Cortex-X925";
static const char ARM_C1Ultra[] = "C1-Ultra";
static const char ARM_N1[] = "Neoverse N1";
static const char ARM_N2[] = "Neoverse N2";
static const char ARM_N3[] = "Neoverse N3";
@@ -63,7 +59,6 @@ namespace ProductNames {
static const char ARM_A65[] = "Cortex-A65";
static const char ARM_A510[] = "Cortex-A510";
static const char ARM_A520[] = "Cortex-A520";
static const char ARM_C1Nano[] = "C1-Nano";
static const char ARM_Kryo200[] = "Kryo 2xx";
static const char ARM_Kryo300[] = "Kryo 3xx";
@@ -75,7 +70,6 @@ namespace ProductNames {
static const char ARM_Denver[] = "Nvidia Denver";
static const char ARM_Carmel[] = "Nvidia Carmel";
static const char ARM_Olympus[] = "Nvidia Olympus";
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
@@ -89,12 +83,8 @@ namespace ProductNames {
static const char ARM_Blizzard_M2Pro[] = "Apple Blizzard (M2 Pro)";
static const char ARM_Avalanche_M2Max[] = "Apple Avalanche (M2 Max)";
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
static const char ARM_AppleSilicon[] = "Apple Silicon";
static const char ARM_ORYON_1[] = "Oryon-1";
static const char ARM_Ampere_1[] = "AmpereOne";
static const char ARM_Ampere_1A[] = "AmpereOneA";
static const char ARM_Ampere_1B[] = "AmpereOneB";
#else
#endif
} // namespace ProductNames
@@ -140,7 +130,7 @@ constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
});
#endif
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
uint32_t GetCycleCounterFrequency() {
uint64_t Result {};
__asm("mrs %[Res], CNTFRQ_EL0" : [Res] "=r"(Result));
@@ -180,7 +170,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
// CPU priority order
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
// Relative list so things they will commonly end up in big.little configurations sort of relate
static constexpr std::array<CPUMIDR, 66> CPUMIDRs = {{
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
// Typically big CPU cores
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
@@ -190,48 +180,39 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x029, 1, ProductNames::ARM_Firestorm_M1Max}, // Apple Firestorm (M1 Max)
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
{0x61, 0, 1, ProductNames::ARM_AppleSilicon}, // QEmu Apple Silicon
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
{0x41, 0xd8b, 1, ProductNames::ARM_C1Pro}, // C1-Pro
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0xc0, 0xac3, 1, ProductNames::ARM_Ampere_1}, // AmpereOne
{0xc0, 0xac4, 1, ProductNames::ARM_Ampere_1A}, // AmpereOneA
{0xc0, 0xac5, 1, ProductNames::ARM_Ampere_1B}, // AmpereOneB
{0x4e, 0x010, 1, ProductNames::ARM_Olympus}, // Olympus
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
// Denver rated above A57 to match TX2 weirdness
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
@@ -246,7 +227,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
{0x41, 0xd8a, 1, ProductNames::ARM_C1Nano}, // C1-Nano
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
@@ -444,10 +424,10 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
Res.eax = FAMILY_IDENTIFIER;
Res.ebx = 0 | // Brand index
(8 << 8) | // Cache line size in bytes
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(GetCPUID() << 24); // Local APIC ID
Res.ebx = 0 | // Brand index
(8 << 8) | // Cache line size in bytes
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(0 << 24); // Local APIC ID
Res.ecx = (1 << 0) | // SSE3
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
@@ -510,7 +490,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
(1 << 25) | // SSE
(1 << 26) | // SSE2
(0 << 27) | // Self Snoop
(0 << 28) | // (HTT) Max APIC IDs reserved field is valid
(1 << 28) | // Max APIC IDs reserved field is valid
(1 << 29) | // Thermal monitor
(0 << 30) | // Reserved
(0 << 31); // Pending break enable
@@ -857,10 +837,10 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0001h(uint32_t Leaf) con
constexpr uint32_t MaximumSubLeafNumber = 2;
if (Leaf == 0) {
// EAX[3:0] Is the host architecture that FEX is running under
#ifdef ARCHITECTURE_x86_64
#ifdef _M_X86_64
// EAX[3:0] = 1 = x86_64 host architecture
Res.eax |= 0b0001;
#elif defined(ARCHITECTURE_arm64)
#elif defined(_M_ARM_64)
// EAX[3:0] = 2 = AArch64 host architecture
Res.eax |= 0b0010;
#else
@@ -912,38 +892,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
Res.eax = FAMILY_IDENTIFIER;
Res.ecx = (1 << 0) | // LAHF/SAHF
(1 << 1) | // 0 = Single core product, 1 = multi core product
(0 << 2) | // SVM
(1 << 3) | // Extended APIC register space
(0 << 4) | // LOCK MOV CR0 means MOV CR8
(1 << 5) | // ABM instructions
(CTX->HostFeatures.SupportsSSE4a << 6) | // SSE4a
(0 << 7) | // Misaligned SSE mode
(1 << 8) | // PREFETCHW
(0 << 9) | // OS visible workaround support
(0 << 10) | // Instruction based sampling support
(0 << 11) | // XOP
(0 << 12) | // SKINIT
(0 << 13) | // Watchdog timer support
(0 << 14) | // Reserved
(0 << 15) | // Lightweight profiling support
(0 << 16) | // FMA4
(1 << 17) | // Translation cache extension
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(0 << 20) | // Reserved
(0 << 21) | // XOP-TBM
(0 << 22) | // Topology extensions support
(0 << 23) | // Core performance counter extensions
(0 << 24) | // NB performance counter extensions
(0 << 25) | // Reserved
(0 << 26) | // Data breakpoints extensions
(0 << 27) | // Performance TSC
(0 << 28) | // L2 perf counter extensions
(0 << 29) | // MONITORX
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.ecx = (1 << 0) | // LAHF/SAHF
(1 << 1) | // 0 = Single core product, 1 = multi core product
(0 << 2) | // SVM
(1 << 3) | // Extended APIC register space
(0 << 4) | // LOCK MOV CR0 means MOV CR8
(1 << 5) | // ABM instructions
(0 << 6) | // SSE4a
(0 << 7) | // Misaligned SSE mode
(1 << 8) | // PREFETCHW
(0 << 9) | // OS visible workaround support
(0 << 10) | // Instruction based sampling support
(0 << 11) | // XOP
(0 << 12) | // SKINIT
(0 << 13) | // Watchdog timer support
(0 << 14) | // Reserved
(0 << 15) | // Lightweight profiling support
(0 << 16) | // FMA4
(1 << 17) | // Translation cache extension
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(0 << 20) | // Reserved
(0 << 21) | // XOP-TBM
(0 << 22) | // Topology extensions support
(0 << 23) | // Core performance counter extensions
(0 << 24) | // NB performance counter extensions
(0 << 25) | // Reserved
(0 << 26) | // Data breakpoints extensions
(0 << 27) | // Performance TSC
(0 << 28) | // L2 perf counter extensions
(0 << 29) | // MONITORX
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.edx = (1 << 0) | // FPU
(1 << 1) | // Virtual mode extensions
@@ -1097,9 +1077,9 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) con
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
uint32_t CoreCount = Cores - 1;
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
(std::bit_ceil(Cores) << 12) | // ApicIdSize: Number of bits in ApicID
(CoreCount << 0); // Count count subtract one
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
(CoreCount << 0); // Count count subtract one
return Res;
}
@@ -1230,7 +1210,7 @@ CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
SetupFeatures();
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
if (SupportsCPUIndexInTPIDRRO) {
GetCPUID = GetCPUID_TPIDRRO;
}
+2 -2
View File
@@ -159,7 +159,7 @@ private:
struct CPUData {
const char* ProductName {};
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
uint32_t MIDR {};
#endif
bool IsBig {};
@@ -277,7 +277,7 @@ private:
// 0: Highest function parameter and ID
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
// 1: Processor info
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
// 2: Cache and TLB info
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
// 3: Serial Number(previously), now reserved
-660
View File
@@ -1,660 +0,0 @@
// SPDX-License-Identifier: MIT
#include "Utils/SpinWaitLock.h"
#include <Interface/Context/Context.h>
#include <Interface/Core/ArchHelpers/Arm64Emitter.h>
#include <Interface/Core/Dispatcher/Dispatcher.h>
#include <Interface/Core/JIT/DebugData.h>
#include <Interface/Core/JIT/Relocations.h>
#include <Interface/Core/LookupCache.h>
#include <Interface/Core/OpcodeDispatcher.h>
#include <Interface/IR/PassManager.h>
#include <FEXCore/Core/Thunks.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXHeaderUtils/Filesystem.h>
#include <git_version.h>
#include <xxhash.h>
#include <fstream>
namespace FEXCore {
#if __clang_major__ < 16
ExecutableFileInfo::ExecutableFileInfo(fextl::unique_ptr<HLE::SourcecodeMap> Map, uint64_t FileId, fextl::string Filename)
: SourcecodeMap(std::move(Map))
, FileId(FileId)
, Filename(Filename) {}
#endif
fextl::string CodeMap::GetBaseFilename(const ExecutableFileInfo& MainExecutable, bool AddNombSuffix) {
auto FileId = MainExecutable.FileId;
std::string_view base_filename = FHU::Filesystem::GetFilename(std::string_view {MainExecutable.Filename});
if (FileId != 0xffff'ffff'ffff'ffff) {
return fextl::fmt::format("{}-{:016x}{}", base_filename, MainExecutable.FileId, AddNombSuffix ? "-nomb" : "");
}
return "";
}
fextl::map<CodeMapFileId, CodeMap::ParsedContents> CodeMap::ParseCodeMap(std::ifstream& File) {
fextl::map<CodeMapFileId, CodeMap::ParsedContents> Ret;
while (true) {
Entry Entry;
File.read(reinterpret_cast<char*>(&Entry), sizeof(Entry));
if (!File) {
break;
}
if (Entry.FileId == LoadExternalLibrary.FileId && Entry.BlockOffset == LoadExternalLibrary.BlockOffset) {
ExternalLibraryInfo Info;
File.read(reinterpret_cast<char*>(&Info), sizeof(Info));
fextl::string Filename;
std::getline(File, Filename, '\0');
// Align to 4-byte boundary
char Null[4];
File.read(Null, AlignUp(Filename.size() + 1, 4) - Filename.size() - 1);
if (!File) {
break;
}
Ret[Info.ExternalFileId].Filename = std::move(Filename);
} else if (Entry.FileId == SetExecutableFileId {}.Marker.FileId && Entry.BlockOffset == SetExecutableFileId {}.Marker.BlockOffset) {
CodeMapFileId ExecutableFileId;
File.read(reinterpret_cast<char*>(&ExecutableFileId), sizeof(ExecutableFileId));
if (!File) {
break;
}
Ret[ExecutableFileId].IsExecutable = true;
} else {
if (!Ret.contains(Entry.FileId)) {
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
} else {
Ret[Entry.FileId].Blocks.insert(Entry.BlockOffset);
}
}
if (!File) {
break;
}
}
return Ret;
}
CodeMapWriter::CodeMapWriter(CodeMapOpener& Opener, bool OpenEagerly)
: Buffer(4096)
, FileOpener(Opener) {
if (OpenEagerly) {
CodeMapFD = FileOpener.OpenCodeMapFile();
}
}
CodeMapWriter::~CodeMapWriter() {
if (CodeMapFD.value_or(-1) != -1) {
Flush(BufferOffset);
close(*CodeMapFD);
}
}
bool CodeMapWriter::IsWriteEnabled(const ExecutableFileSectionInfo& Section) {
if (CodeMapFD == -1) {
return false;
}
// PV libraries can't yet be read by FEXServer, so skip dumping them
if (Section.FileInfo.Filename.starts_with("/run/pressure-vessel")) {
return false;
}
if (CodeMapFD) {
return true;
}
// Acquire mutex and re-check CodeMapFD to avoid race conditions
auto lk = std::unique_lock {Mutex};
if (!CodeMapFD) {
CodeMapFD = FileOpener.OpenCodeMapFile();
}
return CodeMapFD != -1;
}
void CodeMapWriter::Flush(size_t Offset) {
// Acquire exclusive lock and flush circular buffer
std::unique_lock Lock {Mutex};
Flush(Offset, Lock);
}
void CodeMapWriter::Flush(size_t Offset, std::unique_lock<std::shared_mutex>&) {
write(*CodeMapFD, Buffer.data(), Offset);
BufferOffset = 0;
}
void CodeMapWriter::AppendBlock(const FEXCore::ExecutableFileSectionInfo& SectionInfo, uint64_t BlockEntry) {
if (!IsWriteEnabled(SectionInfo)) {
return;
}
BlockEntry -= SectionInfo.FileStartVA;
if (BlockEntry > std::numeric_limits<uint32_t>::max()) {
ERROR_AND_DIE_FMT("Cannot write code map");
}
// Register new library if not already known
bool NewLibraryLoad = false;
{
// Check prior registration with shared lock
std::shared_lock Lock {Mutex};
NewLibraryLoad = !KnownFileIds.contains(SectionInfo.FileInfo.FileId);
}
if (NewLibraryLoad) {
// Register to map with exclusive lock
std::unique_lock Lock {Mutex};
NewLibraryLoad &= KnownFileIds.insert(SectionInfo.FileInfo.FileId).second;
}
if (NewLibraryLoad) {
// Add entry to code map
AppendLibraryLoad(SectionInfo.FileInfo);
}
// Register the actual code block
CodeMap::Entry DataEntry {SectionInfo.FileInfo.FileId, static_cast<uint32_t>(BlockEntry)};
AppendData(std::as_bytes(std::span {&DataEntry, 1}));
}
void CodeMapWriter::AppendLibraryLoad(const FEXCore::ExecutableFileInfo& FileInfo) {
// See CodeMap::ExternalLibraryInfo
auto ExternalFileId = FileInfo.FileId;
auto TotalSize = AlignUp(sizeof(CodeMap::LoadExternalLibrary) + sizeof(ExternalFileId) + FileInfo.Filename.size() + 1, 4);
const auto Data = reinterpret_cast<char*>(alloca(TotalSize));
auto WritePtr = std::copy_n(reinterpret_cast<const char*>(&CodeMap::LoadExternalLibrary), sizeof(CodeMap::LoadExternalLibrary), Data);
WritePtr = std::copy_n(reinterpret_cast<const char*>(&ExternalFileId), sizeof(ExternalFileId), WritePtr);
WritePtr = std::copy(FileInfo.Filename.begin(), FileInfo.Filename.end(), WritePtr);
std::fill(WritePtr, Data + TotalSize, 0);
AppendData(std::as_bytes(std::span {Data, TotalSize}));
}
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo) {
CodeMap::SetExecutableFileId Data {.ExecutableFileId = FileInfo.FileId};
AppendData(std::span {reinterpret_cast<const std::byte*>(&Data), sizeof(Data)});
}
void CodeMapWriter::AppendData(std::span<const std::byte> Data) {
std::shared_lock Lock {Mutex};
auto Offset = BufferOffset.fetch_add(Data.size_bytes());
if (Offset + Data.size_bytes() > Buffer.size()) {
// Acquire exclusive lock and flush the buffer.
// Under heavy pressure, multiple threads may observe an exhausted buffer simultaneously.
// The thread with the last in-bounds Offset is responsible for flushing the buffer.
Lock.unlock();
bool IsResponsibleForFlush = false;
{
std::unique_lock ExclusiveLock {Mutex};
IsResponsibleForFlush = (Offset <= Buffer.size());
if (IsResponsibleForFlush) {
Flush(Offset, ExclusiveLock);
}
}
if (!IsResponsibleForFlush) {
// Wait for the buffer to be flushed on the responsible thread
Utils::SpinWaitLock::WaitPred<std::less_equal<>, size_t>(reinterpret_cast<size_t*>(&BufferOffset), Buffer.size());
}
AppendData(Data);
return;
}
memcpy(&Buffer.at(Offset), Data.data(), Data.size_bytes());
}
} // namespace FEXCore
namespace FEXCore::Context {
CodeCache::CodeCache(ContextImpl& CTX_)
: CTX(CTX_) {}
CodeCache::~CodeCache() = default;
uint64_t CodeCache::ComputeCodeMapId(std::string_view Filename, int FD) {
if (Filename.empty()) {
return 0xffff'ffff'ffff'ffff;
}
// For now, we just use the file path as an identifier.
// TODO: Ensure the hash is unique enough to distinguish executables while remaining independent of the installation location
return XXH3_64bits(Filename.data(), Filename.size());
}
struct CodeCacheHeader {
std::array<char, 4> Magic = ExpectedMagic;
uint32_t FormatVersion = 1;
char FEXVersion[8] = {};
uint32_t NumBlocks;
uint32_t NumCodePages;
uint32_t CodeBufferSize;
uint32_t NumRelocations;
uint64_t SerializedBaseAddress;
// TODO: Consider including information from LookupCache.BlockLinks
static constexpr std::array<char, 4> ExpectedMagic = {'F', 'X', 'C', 'C'};
};
template<typename T>
concept OrderedContainer = requires { typename T::key_compare; };
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
auto CodeBuffer = CTX.GetLatest();
auto& LookupCache = *Thread.LookupCache->Shared;
auto Relocations = Thread.CPUBackend->TakeRelocations(SourceBinary.FileStartVA);
// Write file header
CodeCacheHeader header {};
constexpr std::string_view git_hash = GIT_SHORT_HASH;
static_assert(git_hash.size() <= sizeof(header.FEXVersion));
std::ranges::copy(git_hash, header.FEXVersion);
header.NumBlocks = LookupCache.BlockList.size();
header.NumCodePages = LookupCache.CodePages.size();
header.CodeBufferSize = CTX.LatestOffset;
header.NumRelocations = Relocations.size();
header.SerializedBaseAddress = SerializedBaseAddress;
::write(fd, &header, sizeof(header));
// Dump guest<->host block mappings
{
// Cache contents must be deterministic, so copy the unordered block list and then sort by key
static_assert(!OrderedContainer<decltype(LookupCache.BlockList)>, "Already deterministic; drop temporary container");
fextl::vector<std::pair<uint64_t, const GuestToHostMap::BlockEntry*>> BlockList;
BlockList.reserve(LookupCache.BlockList.size());
for (auto& [Guest, BlockEntry] : LookupCache.BlockList) {
static_assert(sizeof(Guest) == 8, "Breaking change in code cache data layout");
BlockList.emplace_back(Guest, &BlockEntry);
}
std::ranges::sort(BlockList);
for (auto [Guest, Host] : BlockList) {
static_assert(sizeof(Host->HostCode) == 8, "Breaking change in code cache data layout");
static_assert(sizeof(Host->CodePages[0]) == 8, "Breaking change in code cache data layout");
Guest -= SourceBinary.FileStartVA;
::write(fd, &Guest, sizeof(Guest));
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->Ptr);
::write(fd, &HostCode, sizeof(HostCode));
uint64_t NumCodePages = Host->CodePages.size();
::write(fd, &NumCodePages, sizeof(NumCodePages));
LOGMAN_THROW_A_FMT(std::ranges::is_sorted(Host->CodePages), "Code pages aren't sorted");
for (auto CodePage : Host->CodePages) {
CodePage -= SourceBinary.FileStartVA;
::write(fd, &CodePage, sizeof(CodePage));
}
}
}
// Dump relocations
static_assert(sizeof(Relocations[0]) == 48, "Breaking change in code cache data layout");
::write(fd, Relocations.data(), Relocations.size() * sizeof(Relocations[0]));
// Pad to next page in file so that the CodeBuffer can be mmap'ed into process on load
char Zero[64] {};
auto Off = lseek(fd, 0, SEEK_CUR);
while (Off != AlignUp(Off, Utils::FEX_PAGE_SIZE)) {
auto BytesToWrite = std::min(AlignUp(Off, Utils::FEX_PAGE_SIZE) - Off, sizeof(Zero));
::write(fd, Zero, BytesToWrite);
Off += BytesToWrite;
}
// Dump the host code (relocated for position-independent serialization)
std::vector CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->Ptr), reinterpret_cast<std::byte*>(CodeBuffer->Ptr) + CTX.LatestOffset);
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, true)) {
LOGMAN_THROW_A_FMT(false, "Failed to apply code relocations");
return false;
}
::write(fd, CodeBufferData.data(), CodeBufferData.size());
// Dump code pages
static_assert(OrderedContainer<decltype(LookupCache.CodePages)>, "Non-deterministic data source");
for (const auto& [PageIndex, Entrypoints] : LookupCache.CodePages) {
uint64_t PageAddr = (PageIndex << 12) - SourceBinary.FileStartVA;
::write(fd, &PageAddr, sizeof(PageAddr));
uint64_t NumEntrypoints = Entrypoints.size();
::write(fd, &NumEntrypoints, sizeof(NumEntrypoints));
for (uint64_t Entrypoint : Entrypoints) {
Entrypoint -= SourceBinary.FileStartVA;
::write(fd, &Entrypoint, sizeof(Entrypoint));
}
}
return true;
}
bool CodeCache::LoadData(Core::InternalThreadState* Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& BinarySection) {
if (!EnableCodeCaching) {
return true;
}
namespace ranges = std::ranges;
// Read file header
CodeCacheHeader header {};
::memcpy(&header, MappedCacheFile, sizeof(header));
MappedCacheFile += sizeof(header);
LogMan::Msg::IFmt("Cache load: {:5} blocks; base={:#14x}; off={:#9x}-{:#09x}; {:016x} {}", header.NumBlocks, BinarySection.FileStartVA,
BinarySection.BeginVA - BinarySection.FileStartVA, BinarySection.EndVA - BinarySection.FileStartVA,
BinarySection.FileInfo.FileId, BinarySection.FileInfo.Filename);
if (!ranges::equal(header.Magic, header.ExpectedMagic)) {
LogMan::Msg::EFmt("Invalid cache file header");
return false;
}
char ExpectedVersion[8] = GIT_SHORT_HASH;
ranges::fill(ranges::find(ExpectedVersion, 0), std::end(ExpectedVersion), 0);
if (!ranges::equal(header.FEXVersion, ExpectedVersion)) {
LogMan::Msg::IFmt("Cache generated from old FEX version {}, current is {}; skipping", fmt::join(header.FEXVersion, ""),
fmt::join(ExpectedVersion, ""));
return false;
}
if (header.NumBlocks == 0) {
// Valid caches are never empty
LogMan::Msg::IFmt("Code cache empty, aborting");
return false;
}
// Read guest<->host block mappings
using BlockListEntry = decltype(GuestToHostMap::BlockList)::value_type;
fextl::vector<BlockListEntry> BlockList(header.NumBlocks);
{
for (auto& BlockPtr : BlockList) {
::memcpy(&BlockPtr.first, MappedCacheFile, sizeof(BlockPtr.first));
MappedCacheFile += sizeof(BlockPtr.first);
::memcpy(&BlockPtr.second.HostCode, MappedCacheFile, sizeof(BlockPtr.second.HostCode));
MappedCacheFile += sizeof(BlockPtr.second.HostCode);
uint64_t NumGuestPages;
::memcpy(&NumGuestPages, MappedCacheFile, sizeof(NumGuestPages));
MappedCacheFile += sizeof(NumGuestPages);
BlockPtr.second.CodePages.resize(NumGuestPages);
::memcpy(BlockPtr.second.CodePages.data(), MappedCacheFile, std::span {BlockPtr.second.CodePages}.size_bytes());
MappedCacheFile += std::span {BlockPtr.second.CodePages}.size_bytes();
}
// Consistency check: VMA regions at the top and end should belong to the same file
auto [min_val, max_val] = ranges::minmax_element(BlockList, std::less {}, &decltype(BlockList)::value_type::first);
auto MinBound = CTX.SyscallHandler->LookupExecutableFileSection(Thread, min_val->first + BinarySection.FileStartVA);
auto MaxBound = CTX.SyscallHandler->LookupExecutableFileSection(Thread, max_val->first + BinarySection.FileStartVA);
if (&MinBound->FileInfo != &BinarySection.FileInfo || &MaxBound->FileInfo != &BinarySection.FileInfo) {
ERROR_AND_DIE_FMT("Cached blocks offsets {:#x}-{:#x} out of bounds for guest library {} ({:016x} @ {:#x}) while trying to load "
"section {:#x}-{:#x}!",
min_val->first, max_val->first, BinarySection.FileInfo.Filename, BinarySection.FileInfo.FileId,
BinarySection.FileStartVA, BinarySection.BeginVA, BinarySection.EndVA);
}
// Constrain BlockList to the given ExecutableFileSectionInfo
LOGMAN_THROW_A_FMT(ranges::is_sorted(BlockList, [](auto& a, auto& b) { return a.first < b.first; }), "Expected sorted block list");
auto begin = ranges::lower_bound(BlockList, BinarySection.BeginVA - BinarySection.FileStartVA, std::less {}, &BlockListEntry::first);
auto end =
ranges::upper_bound(begin, BlockList.end(), BinarySection.EndVA - BinarySection.FileStartVA - 1, std::less {}, &BlockListEntry::first);
BlockList.erase(end, BlockList.end());
BlockList.erase(BlockList.begin(), begin);
if (BlockList.empty()) {
// Not an error since there is just no data to load
LogMan::Msg::IFmt("No blocks cached in this range, aborting");
return true;
}
}
// Read relocations
fextl::vector<FEXCore::CPU::Relocation> Relocations(header.NumRelocations, FEXCore::CPU::Relocation::Default());
::memcpy(Relocations.data(), MappedCacheFile, Relocations.size() * sizeof(Relocations[0]));
MappedCacheFile += Relocations.size() * sizeof(Relocations[0]);
// Pad to next page in file, which contains CodeBuffer data
MappedCacheFile = reinterpret_cast<std::byte*>(AlignUp(reinterpret_cast<uintptr_t>(MappedCacheFile), Utils::FEX_PAGE_SIZE));
// Prepare CodeBuffer: Page aligned and big enough to hold all cached data
auto Lock = std::unique_lock {CTX.CodeBufferWriteMutex};
if (Thread) {
if (auto Prev = Thread->CPUBackend->CheckCodeBufferUpdate()) {
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
auto lk = Thread->LookupCache->AcquireWriteLock();
Thread->LookupCache->ChangeGuestToHostMapping(*Prev, *CTX.GetLatest()->LookupCache, lk);
}
}
auto CodeBuffer = CTX.GetLatest();
LOGMAN_THROW_A_FMT(header.CodeBufferSize <= CodeBuffer->Size, "CodeBuffer too small to load code cache");
LOGMAN_THROW_A_FMT(reinterpret_cast<uintptr_t>(CodeBuffer->Ptr) % 0x1000 == 0, "Expected CodeBuffer base to be page-aligned");
const auto Delta = AlignUp(CTX.LatestOffset, 0x1000) - CTX.LatestOffset;
CTX.LatestOffset += Delta;
while (CTX.LatestOffset + header.CodeBufferSize > CodeBuffer->Size - Utils::FEX_PAGE_SIZE) {
if (Thread) {
CTX.ClearCodeCache(Thread);
CodeBuffer = CTX.GetLatest();
LogMan::Msg::IFmt("Increased code buffer size to {} MiB for cache load", CodeBuffer->Size / 1024 / 1024);
} else {
ERROR_AND_DIE_FMT("Cannot extend codebuffer without thread!");
}
}
// Read CodeBuffer data from file. Make sure the destination is page-aligned.
// TODO: Only load the data needed for the selected section
auto CodeBufferRange = std::as_writable_bytes(std::span {CodeBuffer->Ptr, CodeBuffer->Size}).subspan(CTX.LatestOffset, header.CodeBufferSize);
::memcpy(CodeBufferRange.data(), MappedCacheFile, header.CodeBufferSize);
MappedCacheFile += header.CodeBufferSize;
CTX.LatestOffset += header.CodeBufferSize;
// Apply FEX relocations
auto Ret = ApplyCodeRelocations(BinarySection.FileStartVA, CodeBufferRange, Relocations, false);
LOGMAN_THROW_A_FMT(Ret == true, "Failed to apply code cache relocations");
{
auto& LookupCache = *CodeBuffer->LookupCache;
auto WriteLock = LookupCache.AcquireWriteLock();
// Register blocks to LookupCache
for (auto& [Guest, Host] : BlockList) {
for (auto& CodePage : Host.CodePages) {
CodePage += BinarySection.FileStartVA;
}
auto HostCode = reinterpret_cast<void*>(Host.HostCode + reinterpret_cast<uintptr_t>(CodeBufferRange.data()));
LookupCache.AddBlockMapping(Guest + BinarySection.FileStartVA, std::move(Host.CodePages), HostCode, WriteLock);
}
// Register loaded code ranges
fextl::vector<uint64_t> Entrypoints;
for (uint32_t i = 0; i < header.NumCodePages; ++i) {
uint64_t CodePage;
memcpy(&CodePage, MappedCacheFile, sizeof(CodePage));
CodePage += BinarySection.FileStartVA;
MappedCacheFile += sizeof(CodePage);
uint64_t NumEntrypoints;
memcpy(&NumEntrypoints, MappedCacheFile, sizeof(NumEntrypoints));
MappedCacheFile += sizeof(NumEntrypoints);
Entrypoints.resize(NumEntrypoints);
memcpy(Entrypoints.data(), MappedCacheFile, NumEntrypoints * sizeof(Entrypoints[0]));
MappedCacheFile += NumEntrypoints * sizeof(Entrypoints[0]);
for (auto& Entrypoint : Entrypoints) {
Entrypoint += BinarySection.FileStartVA;
}
if (LookupCache.AddBlockExecutableRange(Entrypoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE, WriteLock)) {
CTX.SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
}
if (EnableCodeCacheValidation) {
fextl::set<uint64_t> GuestBlocks, HostBlocks;
for (auto& [Guest, Host] : BlockList) {
GuestBlocks.insert(Guest + BinarySection.FileStartVA);
HostBlocks.insert(Host.HostCode);
}
Validate(BinarySection, std::move(GuestBlocks), HostBlocks, CodeBufferRange);
}
return true;
}
void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<uint64_t> GuestBlocks, const fextl::set<uint64_t>& HostBlocks,
std::span<std::byte> CachedCode) {
LOGMAN_THROW_A_FMT(!HostBlocks.empty(), "Tried to validate without any host blocks");
// Skip any cached data before the first host block
CachedCode = CachedCode.subspan(*HostBlocks.begin() - sizeof(CPU::CPUBackend::JITCodeHeader));
if (!ValidationCTX) {
ValidationCTX.reset(static_cast<ContextImpl*>(FEXCore::Context::Context::CreateNewContext(CTX.HostFeatures).release()));
ValidationCTX->SetSignalDelegator(CTX.SignalDelegation);
ValidationCTX->SetSyscallHandler(CTX.SyscallHandler);
ValidationCTX->SetThunkHandler(CTX.ThunkHandler);
if (!ValidationCTX->InitCore()) {
ERROR_AND_DIE_FMT("Failed to create cache load validation context");
}
ValidationThread.reset(ValidationCTX->CreateThread(0, 0, nullptr));
auto Frame = ValidationThread->CurrentFrame;
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_GDT] = &ValidationGDT[0];
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_LDT] = &ValidationGDT[0];
Frame->State.cs_idx = 0;
Frame->State.cs_cached = 0;
if (ValidationCTX->Config.Is64BitMode()) {
ValidationGDT[0].L = 1; // L = Long Mode = 64-bit
ValidationGDT[0].D = 0; // D = Default Operand Size = Reserved
} else {
ValidationGDT[0].L = 0; // L = Long Mode = 32-bit
ValidationGDT[0].D = 1; // D = Default Operand Size = 32-bit
}
}
auto NewCodeBuffer = ValidationCTX->GetLatest();
std::span<std::byte> CodeBufferRangeRef =
std::as_writable_bytes(std::span {NewCodeBuffer->Ptr, NewCodeBuffer->Ptr + NewCodeBuffer->Size}).subspan(0, CachedCode.size_bytes());
while (!GuestBlocks.empty()) {
auto [CompiledBlocks, _, _2, _3, _4] = ValidationCTX->CompileCode(ValidationThread.get(), *GuestBlocks.begin(), 0 /* TODO: Set MaxInst? */);
for (auto& Entry : CompiledBlocks.EntryPoints) {
GuestBlocks.erase(Entry.first);
}
}
// Patch FEX-internal function addresses with values from the main Context to ensure the code blocks are comparable
auto NewRelocations = ValidationThread->CPUBackend->TakeRelocations(Section.FileStartVA);
NewRelocations.erase(std::remove_if(NewRelocations.begin(), NewRelocations.end(), [](const CPU::Relocation& Reloc) {
return Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL && Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
}));
(void)ApplyCodeRelocations(Section.FileStartVA, CodeBufferRangeRef, NewRelocations, false);
if (ValidationCTX->LatestOffset <= CodeBufferRangeRef.size()) {
// Reference compilation produced fewer bytes than our cache, so validation is going to fail.
// Make sure we don't output any garbage bytes though.
CodeBufferRangeRef = CodeBufferRangeRef.subspan(0, ValidationCTX->LatestOffset);
}
auto [Mismatch, _] = std::mismatch(CodeBufferRangeRef.begin(), CodeBufferRangeRef.end(), CachedCode.begin());
if (Mismatch != CodeBufferRangeRef.end()) {
// Align down to instruction size
auto Idx = AlignDown(std::distance(CodeBufferRangeRef.begin(), Mismatch), 4);
auto BlockIt = std::prev(HostBlocks.lower_bound(*HostBlocks.begin() + Idx + 1));
std::optional<uint64_t> GuestBlockAddr;
std::optional<uint64_t> GuestBlockAddrRef;
if (BlockIt != HostBlocks.end()) {
for (int i : {0, 1}) {
std::span Buffer = (i == 0 ? CachedCode : CodeBufferRangeRef);
// Second instruction is always a constant load for relative offset to the (multi)block start
int32_t addr = (*reinterpret_cast<uint32_t*>(&Buffer[*BlockIt - *HostBlocks.begin() + 4]) & 0x3ff'ffe0) << 11;
addr >>= 14;
auto header = reinterpret_cast<CPU::CPUBackend::JITCodeHeader*>(&Buffer[*BlockIt - *HostBlocks.begin() + 4 + addr]);
auto tail = reinterpret_cast<CPU::CPUBackend::JITCodeTail*>(reinterpret_cast<uintptr_t>(header) + header->OffsetToBlockTail);
(i == 0 ? GuestBlockAddr : GuestBlockAddrRef) = tail->RIP - Section.FileStartVA;
LogMan::Msg::EFmt("Recorded rip {}: {:#x} (offset {:#x})", i, tail->RIP, tail->RIP - Section.FileStartVA);
if (i == 1) {
if (tail->RIP >= Section.BeginVA && tail->RIP < Section.EndVA) {
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, _] =
ValidationCTX->GenerateIR(ValidationThread.get(), tail->RIP, false, FEXCore::Config::Get_MAXINST());
fextl::stringstream ss;
FEXCore::IR::Dump(&ss, &*IRView);
LogMan::Msg::EFmt("IR:\n{}", ss.str());
} else {
LogMan::Msg::EFmt("Can't dump IR for out-of-range RIP {:#x}", tail->RIP);
}
}
}
}
fextl::string GuestBlockInfo = "UNKNOWN";
if (GuestBlockAddr) {
GuestBlockInfo = fextl::fmt::format("{:#x}", GuestBlockAddr.value());
}
if (GuestBlockAddr != GuestBlockAddrRef) {
GuestBlockInfo += " (MISMATCH)";
}
ERROR_AND_DIE_FMT("Cache validation failed at offset {:#x}: {:02x} <-> {:02x} (at {} <-> {}, guest block {})", Idx,
fmt::join(CachedCode.subspan(Idx, 4), ""), fmt::join(CodeBufferRangeRef.subspan(Idx, 4), ""),
fmt::ptr(CachedCode.data()), fmt::ptr(CodeBufferRangeRef.data()), GuestBlockInfo);
}
// Reset Context state for next validation
ValidationThread->LookupCache->ClearCache(ValidationThread->LookupCache->AcquireWriteLock());
ValidationCTX->LatestOffset = 0;
LogMan::Msg::IFmt("\tSuccessfully validated cache");
}
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
std::span<const FEXCore::CPU::Relocation> EntryRelocations, bool ForStorage) {
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
Emitter.SetCursorOffset(Reloc.Header.Offset);
switch (Reloc.Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
// Generate a literal so we can place it
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Reloc.NamedSymbolLiteral.Symbol);
Emitter.dc64(Pointer);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Pointers are required to fit within 48-bit VA space.
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer,
CPU::Arm64Emitter::PadType::DOPAD, 6);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
Emitter.dc64(GuestEntry + Reloc.GuestRIP.GuestRIP);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
uint64_t Pointer = Reloc.GuestRIP.GuestRIP + GuestEntry;
// Pointers are required to fit within 48-bit VA space.
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIP.RegisterIndex), Pointer,
CPU::Arm64Emitter::PadType::DOPAD, 6);
break;
}
default: ERROR_AND_DIE_FMT("Unknown relocation type {}", ToUnderlying(Reloc.Header.Type));
}
}
return true;
}
} // namespace FEXCore::Context
+104 -110
View File
@@ -18,7 +18,6 @@ $end_info$
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <Interface/GDBJIT/GDBJIT.h>
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
@@ -78,7 +77,7 @@ namespace FEXCore::Context {
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
: HostFeatures {Features}
, CPUID {this}
, CodeCache {*this} {
, IRCaptureCache {this} {
if (!Config.Is64BitMode()) {
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
Config.VirtualMemSize = 1ULL << 32;
@@ -345,7 +344,7 @@ bool ContextImpl::InitCore() {
// Set up the SignalDelegator config since core is initialized.
SignalDelegation->SetConfig(Dispatcher->MakeSignalDelegatorConfig());
#if defined(_WIN32) && !defined(ARCHITECTURE_arm64ec)
#if defined(_WIN32) && !defined(_M_ARM_64EC)
// WOW64 always needs the interrupt fault check to be enabled.
Config.NeedsPendingInterruptFaultCheck = true;
#endif
@@ -363,9 +362,6 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
}
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
// Update the thread pointer for Thunk return to the latest.
Thread->CurrentFrame->Pointers.ThunkCallbackRet = SignalDelegation->GetThunkCallbackRET();
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
// If it is the parent thread that died then just leave
@@ -379,10 +375,8 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
Thread->CurrentFrame->State.L1Pointer = Thread->LookupCache->GetL1Pointer();
Thread->CurrentFrame->State.L1Mask = Thread->LookupCache->GetScaledL1PointerMask();
Thread->CurrentFrame->Pointers.L2Pointer = Thread->LookupCache->GetPagePointer();
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
Dispatcher->InitThreadPointers(Thread);
@@ -403,7 +397,6 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXC
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
.CTX = this,
};
FEXCore::Allocator::VirtualName("FEXMem_ThreadState", Thread, sizeof(*Thread));
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
Thread->CurrentFrame->State.rip = InitialRIP;
@@ -440,10 +433,6 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
Profiler::PostForkAction(Child);
if (Child) {
if (CodeMapWriter) {
CodeMapWriter->ResetAfterFork();
}
CodeInvalidationMutex.StealAndDropActiveLocks();
if (Config.StrictInProcessSplitLocks) {
StrictSplitLockMutex = 0;
@@ -466,14 +455,9 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
}
#endif
void ContextImpl::OnCodeBufferAllocated(const fextl::shared_ptr<CPU::CodeBuffer>& Buffer) {
void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
if (Config.GlobalJITNaming()) {
Symbols.RegisterJITSpace(Buffer->Ptr, Buffer->Size);
}
{
std::scoped_lock lk {CodeBufferListLock};
CodeBufferList.emplace_back(Buffer);
Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
}
@@ -485,8 +469,7 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, boo
Thread->CPUBackend->ClearCache();
} else {
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
auto lk = Thread->LookupCache->AcquireWriteLock();
Thread->LookupCache->ClearCache(lk);
Thread->LookupCache->ClearCache();
}
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
}
@@ -627,7 +610,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
} else {
ForceTSO = IR::ForceTSOMode::ForceDisabled;
}
} else if (DecodedInfo->Flags & X86Tables::DecodeFlags::FLAG_FORCE_TSO) {
} else if (DecodedInfo->ForceTSO) {
ForceTSO = IR::ForceTSOMode::ForceEnabled;
}
@@ -658,11 +641,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
}
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::INVALID_INST ||
Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::BAD_RELOCATION) {
Thread->OpDispatcher->InvalidOp(DecodedInfo);
} else {
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::NOEXEC_INST) {
Thread->OpDispatcher->NoExecOp(DecodedInfo);
} else {
Thread->OpDispatcher->InvalidOp(DecodedInfo);
}
}
@@ -726,10 +708,9 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
if (SourcecodeResolver && Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
if (MappedSection) {
MappedSection->FileInfo.SourcecodeMap =
SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, CodeMap::GetBaseFilename(MappedSection->FileInfo, false));
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (AOTIRCacheEntry.Entry) {
AOTIRCacheEntry.Entry->SourcecodeMap = SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
}
}
@@ -746,7 +727,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
// but this would increase lock contention. Redundant frontend runs aren't
// as expensive and are easily reverted.
if (MaxInst != 1) {
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
Thread->OpDispatcher->DelayedDisownBuffer();
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
.DebugData = nullptr,
@@ -787,13 +768,10 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
// Is the code in the cache?
// The backends only check L1 and L2, not L3
if (auto HostCode = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
if (auto HostCode = Thread->LookupCache->FindBlock(GuestRIP)) {
return HostCode;
}
// Accumulate a JIT count now, as even if another thread raced us, it should count as a compile.
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedJITCount, 1);
auto [CompiledCode, DebugData, StartAddr, Length, NeedsAddGuestCodeRanges] = CompileCode(Thread, GuestRIP, MaxInst);
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
if (CodePtr == nullptr) {
@@ -807,72 +785,52 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
if (Config.BlockJITNaming()) {
auto FragmentBasePtr = CompiledCode.BlockBegin;
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
if (DebugData) {
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
}
}
}
} else {
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
}
}
}
if (Config.LibraryJITNaming() || Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
if (MappedSection) {
if (Config.LibraryJITNaming()) {
Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, MappedSection->FileInfo.Filename);
}
if (Config.GDBSymbols()) {
GDBJITRegister(MappedSection->FileInfo, MappedSection->FileStartVA, GuestRIP, (uintptr_t)CodePtr, *DebugData);
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
}
}
}
}
// Clear any relocations that might have been generated
if (!CodeCache.IsGeneratingCache) {
Thread->CPUBackend->ClearRelocations();
}
Thread->CPUBackend->ClearRelocations();
fextl::vector<uint64_t> CodePages;
if (IRCaptureCache.PostCompileCode(Thread, CompiledCode.BlockBegin, GuestRIP, StartAddr, Length, DebugData.get())) {
// Early exit
return (uintptr_t)CodePtr;
}
if (NeedsAddGuestCodeRanges) {
// Track in the guest to host map all entrypoints for all pages the compiled block touches, if any page didn't previously
// contain code, inform the frontend so it can setup SMC detection.
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
CodePages.reserve(BlockInfo->CodePages.size());
CodePages.insert(CodePages.end(), BlockInfo->CodePages.begin(), BlockInfo->CodePages.end());
for (auto CodePage : BlockInfo->CodePages) {
if (Thread->LookupCache->AddBlockExecutableRange(Thread, BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
if (Thread->LookupCache->AddBlockExecutableRange(BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
}
// Insert to lookup cache
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
Thread->LookupCache->AddBlockMapping(Thread, GuestAddr, CodePages, HostAddr);
}
if (CodeMapWriter) {
auto Region = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
if (Region && Region->FileStartVA != 0) {
CodeMapWriter->AppendBlock(*Region, GuestRIP);
}
Thread->LookupCache->AddBlockMapping(GuestAddr, HostAddr);
}
return (uintptr_t)CodePtr;
@@ -899,37 +857,66 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
return (uintptr_t)CodePtr;
}
void ContextImpl::InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) {
FEXCORE_PROFILE_SCOPED("InvalidateCodeBuffersCodeRange");
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
std::scoped_lock lk {CodeBufferListLock};
auto it = CodeBufferList.begin();
while (it != CodeBufferList.end()) {
if (auto Strong = it->lock()) {
Strong->LookupCache->InvalidateRange(Start, Length);
it++;
} else {
it = CodeBufferList.erase(it);
}
}
}
void ContextImpl::InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
uint64_t Start, uint64_t Length) {
// Ensures now-modified mappings aren't cached as being in their previous non-executable state.
// Accessing FrontendDecoder is safe as the thread's code invalidation mutex must be locked here.
Thread->FrontendDecoder->ResetExecutableRangeCache();
if (Thread->LookupCache->InvalidateCacheRange(Start, Length)) {
FEXCORE_PROFILE_SCOPED("InvalidateCallRet");
auto lk = Thread->LookupCache->AcquireLock();
auto& CodePages = Thread->LookupCache->Shared->CodePages;
auto lower = CodePages.lower_bound(Start >> 12);
auto upper = CodePages.upper_bound((Start + Length - 1) >> 12);
for (auto it = lower; it != upper; it++) {
Accumulator.emplace_back(std::move(it->second));
}
bool InvalidatedAnyEntries = false;
for (const auto& PageEntries : Accumulator) {
for (const auto& Entry : PageEntries) {
if (ContextImpl::ThreadRemoveCodeEntry(Thread, Entry)) {
InvalidatedAnyEntries = true;
}
}
}
if (InvalidatedAnyEntries) {
// This may cause access violations in the thread on Windows as zeroing is not atomic, this is handled by the frontend
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
}
}
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
uint64_t Start, uint64_t Length) {
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
}
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
if (!Thread) {
return;
}
if (!IsMemoryShared) {
IsMemoryShared = true;
UpdateAtomicTSOEmulationConfig();
if (Config.TSOAutoMigration) {
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
Thread->LookupCache->ClearCache();
}
}
}
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
"be unique_locked here");
return Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
}
void ContextImpl::ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
static_cast<ContextImpl*>(Frame->Thread->CTX)->SyscallHandler->InvalidateGuestCodeRange(Frame->Thread, GuestRIP, 1);
}
@@ -971,13 +958,12 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
const auto GPRSize = this->Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
// Thunk entry-points don't get cached, don't need to be padded.
if (GPRSize == IR::OpSize::i64Bit) {
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
R->Reg = IR::PhysicalRegister(IR::RegClass::GPRFixed, X86State::REG_R11).Raw;
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
} else {
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
offsetof(Core::CPUState, mm[0][0]));
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
offsetof(Core::CPUState, mm[0][0]));
}
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
},
@@ -998,7 +984,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
void ContextImpl::AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) {
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
ForceTSOValidRanges.Insert(ValidRanges);
ForceTSOInstructions.merge(std::move(Instructions));
ForceTSOInstructions.merge(Instructions);
}
void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
@@ -1029,9 +1015,9 @@ void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint
auto lk = GuardSignalDeferringSection(CTX->CodeInvalidationMutex, Thread);
if (Size == 8) {
*reinterpret_cast<uint64_t*>(Address) = Value;
*reinterpret_cast<uint64_t *>(Address) = Value;
} else if (Size == 4) {
*reinterpret_cast<uint32_t*>(Address) = Value;
*reinterpret_cast<uint32_t *>(Address) = Value;
} else {
ERROR_AND_DIE_FMT("Unexpected write size for backpatcher: {}", Size);
}
@@ -1040,7 +1026,15 @@ void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint
CTX->SyscallHandler->InvalidateGuestCodeRange(Thread, Address, Size);
}
IR::AOTIRCacheEntry* ContextImpl::LoadAOTIRCacheEntry(const fextl::string& filename) {
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
return rv;
}
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {}
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
}
} // namespace FEXCore::Context
@@ -1,10 +1,11 @@
// SPDX-License-Identifier: MIT
#include "Common/VectorRegType.h"
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/X86HelperGen.h"
#include "Utils/MemberFunctionToPointer.h"
#include <FEXCore/Config/Config.h>
@@ -16,7 +17,6 @@
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <CodeEmitter/Emitter.h>
@@ -25,7 +25,9 @@
#endif
#include <array>
#include <atomic>
#include <bit>
#include <condition_variable>
#include <csignal>
#include <cstring>
@@ -35,14 +37,12 @@ static void SleepThread(FEXCore::Context::ContextImpl* CTX, FEXCore::Core::CpuSt
CTX->SyscallHandler->SleepThread(CTX, Frame);
}
constexpr size_t MAX_DISPATCHER_CODE_SIZE = FEXCore::Utils::FEX_PAGE_SIZE * 4;
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 4;
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl* ctx)
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
, CTX {ctx} {
EmitDispatcher();
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(GetBufferBase()), MAX_DISPATCHER_CODE_SIZE);
}
Dispatcher::~Dispatcher() {
@@ -92,12 +92,12 @@ void Dispatcher::EmitDispatcher() {
FillStaticRegs();
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
ARMEmitter::BiDirectionalLabel LoopTop {};
#ifdef ARCHITECTURE_arm64ec
(void)b(&LoopTop);
#ifdef _M_ARM_64EC
b(&LoopTop);
AbsoluteLoopTopAddressEnterECFillSRA = GetCursorAddress<uint64_t>();
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
@@ -105,10 +105,10 @@ void Dispatcher::EmitDispatcher() {
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
// Force a single instruction block if ENTRY_FILL_SRA_SINGLE_INST_REG is nonzero entering the JIT, used for inline SMC handling.
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
// Enter JIT
(void)b(&LoopTop);
b(&LoopTop);
AbsoluteLoopTopAddressEnterEC = GetCursorAddress<uint64_t>();
// Load ThreadState and write the target PC there
@@ -129,7 +129,7 @@ void Dispatcher::EmitDispatcher() {
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, REG_CALLRET_SP);
// EC_CALL_CHECKER_PC_REG is REG_PF which isn't touched by any of the above
sub(ARMEmitter::Size::i64Bit, TMP1, EC_CALL_CHECKER_PC_REG, TMP1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
// If the entry at the TOS is for the target address, pop it and return to the JIT code
add(ARMEmitter::Size::i64Bit, REG_CALLRET_SP, REG_CALLRET_SP, 0x10);
@@ -141,13 +141,13 @@ void Dispatcher::EmitDispatcher() {
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
(void)Bind(&LoopTop);
Bind(&LoopTop);
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
// Load in our RIP
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
// Clobbers TMP1/2
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
ARMEmitter::ForwardLabel l_NotECCode;
@@ -159,82 +159,75 @@ void Dispatcher::EmitDispatcher() {
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
(void)tbz(TMP1, 0, &l_NotECCode);
tbz(TMP1, 0, &l_NotECCode);
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
mov(EC_CALL_CHECKER_PC_REG, RipReg);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionEC));
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
(void)Bind(&l_NotECCode);
Bind(&l_NotECCode);
#endif
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
}
ARMEmitter::ForwardLabel NoBlock;
if (DisableL2Cache()) {
(void)b(&NoBlock);
} else {
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.L2Pointer));
{
// Offset the address and add to our page pointer
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
}
// Load the pointer from the offset
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
// If page pointer is zero then we have no block
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
// Shift the offset by the size of the block cache entry
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
// The the full LookupCacheEntry with a single LDP.
// Check the guest address first to ensure it maps to the address we are currently at.
// This fixes aliasing problems
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
// If the guest address doesn't match, Compile the block.
sub(TMP2, TMP2, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
// Check the host address to see if it matches, else compile the block.
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
// If we've made it here then we have a real compiled block
{
// Offset the address and add to our page pointer
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
// update L1 cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
// Load the pointer from the offset
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
// If page pointer is zero then we have no block
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
// Shift the offset by the size of the block cache entry
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
// The the full LookupCacheEntry with a single LDP.
// Check the guest address first to ensure it maps to the address we are currently at.
// This fixes aliasing problems
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
// If the guest address doesn't match, Compile the block.
sub(TMP2, TMP2, RipReg);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
// Check the host address to see if it matches, else compile the block.
(void)cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
// L1Mask is pre-shifted.
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg.R(), ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
add(TMP1, TMP1, TMP2);
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
// Jump to the block
br(TMP4);
}
// Jump to the block
br(TMP4);
}
}
@@ -259,7 +252,7 @@ void Dispatcher::EmitDispatcher() {
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
#endif
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
@@ -267,7 +260,7 @@ void Dispatcher::EmitDispatcher() {
Body();
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
strb(ARMEmitter::WReg::zr, TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
#endif
@@ -291,7 +284,7 @@ void Dispatcher::EmitDispatcher() {
mov(ARMEmitter::XReg::x0, STATE);
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionLink));
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
} else {
@@ -310,7 +303,7 @@ void Dispatcher::EmitDispatcher() {
// Need to create the block
{
(void)Bind(&NoBlock);
Bind(&NoBlock);
EmitSignalGuardedRegion([&]() {
SpillStaticRegs(TMP1);
@@ -344,7 +337,7 @@ void Dispatcher::EmitDispatcher() {
}
{
(void)Bind(&CompileSingleStep);
Bind(&CompileSingleStep);
EmitSignalGuardedRegion([&]() {
SpillStaticRegs(TMP1);
@@ -488,7 +481,7 @@ void Dispatcher::EmitDispatcher() {
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.ThunkCallbackRet));
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->X86CodeGen.CallbackReturn);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, CTX->Config.Is64BitMode ? 16 : 12);
@@ -506,7 +499,7 @@ void Dispatcher::EmitDispatcher() {
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
// Now go back to the regular dispatcher loop
(void)b(&LoopTop);
b(&LoopTop);
}
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
@@ -544,8 +537,8 @@ void Dispatcher::EmitDispatcher() {
return Address;
};
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.LUDIV));
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.LDIV));
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
// Interpreter fallbacks
{
@@ -575,15 +568,14 @@ void Dispatcher::EmitDispatcher() {
}
}
(void)Bind(&l_CTX);
Bind(&l_CTX);
dc64(reinterpret_cast<uintptr_t>(CTX));
(void)Bind(&l_Sleep);
Bind(&l_Sleep);
dc64(reinterpret_cast<uint64_t>(SleepThread));
(void)Bind(&l_CompileBlock);
Bind(&l_CompileBlock);
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
dc64(PMFCompileBlock.GetConvertedPointer());
(void)Bind(&l_CompileSingleStep);
Bind(&l_CompileSingleStep);
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
dc64(PMFCompileSingleStep.GetConvertedPointer());
@@ -1086,25 +1078,27 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto& Ptrs = Thread->CurrentFrame->Pointers;
auto& Common = Thread->CurrentFrame->Pointers.Common;
Ptrs.DispatcherLoopTop = AbsoluteLoopTopAddress;
Ptrs.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Ptrs.DispatcherLoopTopEnterEC = AbsoluteLoopTopAddressEnterEC;
Ptrs.DispatcherLoopTopEnterECFillSRA = AbsoluteLoopTopAddressEnterECFillSRA;
Ptrs.ExitFunctionLinker = ExitFunctionLinkerAddress;
Ptrs.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
Ptrs.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
Ptrs.GuestSignal_SIGILL = GuestSignal_SIGILL;
Ptrs.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
Ptrs.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
Ptrs.SignalReturnHandler = SignalHandlerReturnAddress;
Ptrs.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
Ptrs.LUDIVHandler = LUDIVHandlerAddress;
Ptrs.LDIVHandler = LDIVHandlerAddress;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Common.DispatcherLoopTopEnterEC = AbsoluteLoopTopAddressEnterEC;
Common.DispatcherLoopTopEnterECFillSRA = AbsoluteLoopTopAddressEnterECFillSRA;
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
Common.SignalReturnHandler = SignalHandlerReturnAddress;
Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
auto& AArch64 = Thread->CurrentFrame->Pointers.AArch64;
AArch64.LUDIVHandler = LUDIVHandlerAddress;
AArch64.LDIVHandler = LDIVHandlerAddress;
// Fill in the fallback handlers
InterpreterOps::FillFallbackIndexPointers(Ptrs.FallbackHandlerPointers, &ABIPointers[0]);
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers, &ABIPointers[0]);
}
}
@@ -4,7 +4,6 @@
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/memory.h>
#include <array>
@@ -51,10 +50,6 @@ public:
}
#endif
uint64_t GetExitFunctionLinkerAddress() const {
return ExitFunctionLinkerAddress;
}
SignalDelegatorConfig MakeSignalDelegatorConfig() const;
protected:
@@ -97,8 +92,6 @@ private:
void EmitDispatcher();
uint64_t GenerateABICall(FallbackABI ABI);
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
};
} // namespace FEXCore::CPU
+93 -138
View File
@@ -9,6 +9,7 @@ $end_info$
#include "Interface/Context/Context.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Core/LookupCache.h"
#include <array>
@@ -89,6 +90,11 @@ Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
}
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
// Treat FEX-internal X86 callbacks as always executable
if (EntryPoint == CTX->X86CodeGen.CallbackReturn) {
return true;
}
while (Address < ExecutableRangeBase || Address + Size > ExecutableRangeEnd) {
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, Address);
ExecutableRangeBase = RangeInfo.Base;
@@ -132,7 +138,7 @@ std::optional<uint8_t> Decoder::PeekByte(uint8_t Offset) {
}
}
std::pair<uint64_t, bool> Decoder::ReadData(uint8_t Size) {
uint64_t Decoder::ReadData(uint8_t Size) {
LOGMAN_THROW_A_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
uint64_t Res = 0;
@@ -154,21 +160,7 @@ std::pair<uint64_t, bool> Decoder::ReadData(uint8_t Size) {
SkipBytes(Size);
#endif
if (Relocations) {
uint32_t SectionOffset = static_cast<uint32_t>(Address - SectionMinAddress);
if (auto It = Relocations->find(SectionOffset); It != Relocations->end()) {
if (It->second == GuestRelocationType::Rel32 && Size == 4) {
return {static_cast<int64_t>(static_cast<int32_t>(Res) - static_cast<int32_t>(EntryPoint)), true};
} else if (It->second == GuestRelocationType::Rel64 && Size == 8) {
return {static_cast<int64_t>(Res) - static_cast<int64_t>(EntryPoint), true};
} else {
HitBadRelocation = true;
Res = 0;
}
}
}
return {Res, false};
return Res;
}
void Decoder::DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM) {
@@ -200,9 +192,7 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModR
DisplacementSize = 1;
}
if (DisplacementSize) {
bool IsRelocation = false;
std::tie(Literal, IsRelocation) = ReadData(DisplacementSize);
LOGMAN_THROW_A_FMT(!IsRelocation, "1/2 byte relocations unsupported");
Literal = ReadData(DisplacementSize);
if (DisplacementSize == 1) {
Literal = static_cast<int8_t>(Literal);
}
@@ -269,13 +259,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
if (HasSIB) {
FEXCore::X86Tables::SIBDecoded SIB;
if (DecodeInst->Flags & DecodeFlags::FLAG_DECODED_SIB) {
if (DecodeInst->DecodedSIB) {
SIB.Hex = DecodeInst->SIB;
} else {
// Haven't yet grabbed SIB, pull it now
DecodeInst->SIB = ReadByte();
SIB.Hex = DecodeInst->SIB;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_SIB;
DecodeInst->DecodedSIB = true;
}
// If the SIB base is 0b101, aka BP or R13 then we have a 32bit displacement
@@ -308,10 +298,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
LOGMAN_THROW_A_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
if (Displacement) {
auto [Literal, IsRelocation] = ReadData(Displacement);
if (IsRelocation) {
Operand->Type = DecodedOperand::OpType::SIBRelocation;
}
uint64_t Literal = ReadData(Displacement);
if (Displacement == 1) {
Literal = static_cast<int8_t>(Literal);
}
@@ -321,9 +308,10 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
if (ModRM.rm == 0b101) {
// 32bit Displacement
auto [Literal, IsRelocation] = ReadData(4);
Operand->Type = IsRelocation ? DecodedOperand::OpType::RIPRelativeRelocation : DecodedOperand::OpType::RIPRelative;
Operand->Data.RIPLiteral.Value = Literal;
const uint32_t Literal = ReadData(4);
Operand->Type = DecodedOperand::OpType::RIPRelative;
Operand->Data.RIPLiteral.Value.u = Literal;
} else {
// Register-direct addressing
Operand->Type = DecodedOperand::OpType::GPRDirect;
@@ -331,12 +319,12 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
}
} else {
uint8_t DisplacementSize = ModRM.mod == 1 ? 1 : 4;
auto [Literal, IsRelocation] = ReadData(DisplacementSize);
uint32_t Literal = ReadData(DisplacementSize);
if (DisplacementSize == 1) {
Literal = static_cast<int8_t>(Literal);
}
Operand->Type = IsRelocation ? DecodedOperand::OpType::GPRIndirectRelocation : DecodedOperand::OpType::GPRIndirect;
Operand->Type = DecodedOperand::OpType::GPRIndirect;
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
Operand->Data.GPRIndirect.Displacement = Literal;
}
@@ -413,9 +401,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// If we require ModRM and haven't decoded it yet, do it now
// Some instructions have to read modrm upfront, others do it later
if (HasMODRM && !(DecodeInst->Flags & DecodeFlags::FLAG_DECODED_MODRM)) {
if (HasMODRM && !DecodeInst->DecodedModRM) {
DecodeInst->ModRM = ReadByte();
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
}
// New instruction size decoding
@@ -448,8 +436,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_16BIT);
DestSize = 2;
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) && (HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) &&
(HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_64BIT);
DestSize = 8;
} else {
@@ -476,8 +465,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// See table 1-2. Operand-Size Overrides for this decoding
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) && (HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) &&
(HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_64BIT);
} else {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_32BIT);
@@ -632,32 +622,25 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
if (Bytes != 0) {
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
auto [Literal, IsRelocation] = ReadData(Bytes);
if (IsRelocation) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::LiteralRelocation;
DecodeInst->Src[CurrentSrc].Data.LiteralRelocation.EntrypointOffset = Literal;
} else {
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
uint64_t Literal = ReadData(Bytes);
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT) ||
(DecodeFlags::GetSizeDstFlags(DecodeInst->Flags) == DecodeFlags::SIZE_64BIT &&
Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT64BIT)) {
if (Bytes == 1) {
Literal = static_cast<int8_t>(Literal);
} else if (Bytes == 2) {
Literal = static_cast<int16_t>(Literal);
} else {
Literal = static_cast<int32_t>(Literal);
}
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT) || (DecodeFlags::GetSizeDstFlags(DecodeInst->Flags) == DecodeFlags::SIZE_64BIT &&
Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT64BIT)) {
if (Bytes == 1) {
Literal = static_cast<int8_t>(Literal);
} else if (Bytes == 2) {
Literal = static_cast<int16_t>(Literal);
} else {
Literal = static_cast<int32_t>(Literal);
}
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
}
Bytes = 0;
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
}
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
@@ -689,7 +672,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
} else if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -706,18 +689,18 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
constexpr uint16_t PF_F2 = 3;
uint16_t PrefixType = PF_NONE;
if (LastEscapePrefix == 0xF3) {
if (DecodeInst->LastEscapePrefix == 0xF3) {
PrefixType = PF_F3;
} else if (LastEscapePrefix == 0xF2) {
} else if (DecodeInst->LastEscapePrefix == 0xF2) {
PrefixType = PF_F2;
} else if (LastEscapePrefix == 0x66) {
} else if (DecodeInst->LastEscapePrefix == 0x66) {
PrefixType = PF_66;
}
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -744,7 +727,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
return NormalOp(&(*X87Table)[X87Op], X87Op);
@@ -806,7 +789,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -830,7 +813,6 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
InstructionSize = 0;
LastEscapePrefix = 0;
Instruction.fill(0);
DecodeInst = &DecodedBuffer[DecodedSize];
@@ -848,12 +830,11 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
switch (EscapeOp) {
case 0x0F:
[[unlikely]] { // 3DNow!
DecodeREXIfValid(-2);
// 3DNow! Instruction Encoding: 0F 0F [ModRM] [SIB] [Displacement] [Opcode]
// Decode ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -873,7 +854,6 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
break;
}
case 0x38: { // F38 Table!
DecodeREXIfValid(-2);
constexpr uint16_t PF_38_NONE = 0;
constexpr uint16_t PF_38_66 = (1U << 0);
constexpr uint16_t PF_38_F2 = (1U << 1);
@@ -893,23 +873,23 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
uint16_t LocalOp = (Prefix << 8) | ReadByte();
bool NoOverlay66 = (FEXCore::X86Tables::H0F38TableOps[LocalOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
if (LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
if (DecodeInst->LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather than modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
DecodeFlags::PopOpAddrIf(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
}
return NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp);
break;
}
case 0x3A: { // F3A Table!
DecodeREXIfValid(-2);
constexpr uint16_t PF_3A_NONE = 0;
constexpr uint16_t PF_3A_66 = (1 << 0);
constexpr uint16_t PF_3A_REX = (1 << 1);
uint16_t Prefix = PF_3A_NONE;
if (LastEscapePrefix == 0x66) { // Operand Size
if (DecodeInst->LastEscapePrefix == 0x66) { // Operand Size
Prefix = PF_3A_66;
}
@@ -933,20 +913,19 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
bool NoOverlay = (FEXCore::X86Tables::SecondBaseOps[EscapeOp].Flags & InstFlags::FLAGS_NO_OVERLAY) != 0;
bool NoOverlay66 = (FEXCore::X86Tables::SecondBaseOps[EscapeOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
DecodeREXIfValid(-2);
if (NoOverlay) { // This section of the table ignores prefix extention
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0xF3) { // REP
} else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0xF2) { // REPNE
} else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
} else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
@@ -962,7 +941,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
}
case 0x66: // Operand Size prefix
DecodeInst->Flags |= DecodeFlags::FLAG_OPERAND_SIZE;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
break;
case 0x67: // Address Size override prefix
@@ -993,11 +972,11 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
break;
case 0xF2: // REPNE prefix
DecodeInst->Flags |= DecodeFlags::FLAG_REPNE_PREFIX;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
break;
case 0xF3: // REP prefix
DecodeInst->Flags |= DecodeFlags::FLAG_REP_PREFIX;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
break;
case 0x64: // FS prefix
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_FS_PREFIX;
@@ -1013,9 +992,29 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
}
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
DecodeInst->REXIndex = InstructionSize;
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
// Widening displacement
if (Op & 0b1000) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_WIDENING;
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_WIDENING_SIZE_LAST);
}
// XGPR_B bit set
if (Op & 0b0001) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
}
// XGPR_X bit set
if (Op & 0b0010) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
}
// XGPR_R bit set
if (Op & 0b0100) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
}
} else {
DecodeREXIfValid();
return NormalOpHeader(Info, Op);
}
@@ -1031,53 +1030,17 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
return true;
}
void Decoder::DecodeREXIfValid(int8_t ExpectedOffset) {
LOGMAN_THROW_A_FMT(ExpectedOffset < 0, "Expecting an negative offset for the REX offset!");
const int8_t REXIndex = InstructionSize + ExpectedOffset;
if (DecodeInst->REXIndex != 0 && DecodeInst->REXIndex == REXIndex) {
const uint8_t Op = Instruction[REXIndex - 1];
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
// Widening displacement
if (Op & 0b1000) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_WIDENING;
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_WIDENING_SIZE_LAST);
}
// XGPR_B bit set
if (Op & 0b0001) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
}
// XGPR_X bit set
if (Op & 0b0010) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
}
// XGPR_R bit set
if (Op & 0b0100) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
}
}
}
Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
// Will be set if DecodeInstructionImpl tries to read non-executable memory
HitNonExecutableRange = false;
HitBadRelocation = false;
bool ErrorDuringDecoding = !DecodeInstructionImpl(PC);
if (ErrorDuringDecoding || HitNonExecutableRange || HitBadRelocation) [[unlikely]] {
if (ErrorDuringDecoding || HitNonExecutableRange) [[unlikely]] {
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
// Error while decoding instruction. We don't know the table or instruction size
DecodeInst->TableInfo = nullptr;
auto Result = ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST :
DecodeInst->InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST :
HitNonExecutableRange ? DecodedBlockStatus::NOEXEC_INST :
DecodedBlockStatus::BAD_RELOCATION;
DecodeInst->InstSize = 0;
return Result;
return ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST : DecodedBlockStatus::NOEXEC_INST;
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
return DecodedBlockStatus::INVALID_INST;
@@ -1092,10 +1055,10 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
if (DecodeInst->OP == 0x8b && DecodeInst->Src[0].IsGPRIndirect() &&
IsKnownAtomicDisplacement(DecodeInst->Src[0].Data.GPRIndirect.Displacement)) {
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
DecodeInst->ForceTSO = true;
}
if (DecodeInst->OP == 0x89 && DecodeInst->Dest.IsGPRIndirect() && IsKnownAtomicDisplacement(DecodeInst->Dest.Data.GPRIndirect.Displacement)) {
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
DecodeInst->ForceTSO = true;
}
}
@@ -1161,9 +1124,9 @@ void Decoder::BranchTargetInMultiblockRange() {
// Forbid distant branches to have the cost code better match the guest code layout, avoiding massive (range-wise) code
// blocks in highly fragmented guest code. Such branches are often not-taken branches to garbage in obfuscated code.
constexpr uint64_t MAX_FORWARD_BRANCH_DIST = FEXCore::Utils::FEX_PAGE_SIZE * 4;
bool ValidMultiblockMember = TargetRIP >= EntryPoint && TargetRIP < std::min(InstEnd + MAX_FORWARD_BRANCH_DIST, SectionMaxAddress);
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < std::min(InstEnd + MAX_FORWARD_BRANCH_DIST, SymbolMaxAddress);
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
#endif
@@ -1338,7 +1301,7 @@ const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, u
return _InstStream - EntryPoint + RIP;
}
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
BlockInfo.TotalInstructionCount = 0;
BlockInfo.Blocks.clear();
@@ -1354,23 +1317,19 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
BlockInfo.Is64BitMode = CSSegment->L == 1;
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
// XXX: Load symbol data
SymbolAvailable = false;
EntryPoint = PC;
BlockInfo.EntryPoints = {PC};
InstStream = _InstStream;
uint64_t TotalInstructions {};
SectionMinAddress = 0;
SectionMaxAddress = ~0ULL;
Relocations = nullptr;
if (CTX->GetCodeCache().IsGeneratingCache || EnableCodeCacheValidation) {
// If generating cache, attempt to load section bounds and relocations
if (auto SectionInfo = CTX->SyscallHandler->LookupExecutableFileSection(Thread, EntryPoint)) {
SectionMinAddress = SectionInfo->FileStartVA;
SectionMaxAddress = SectionInfo->EndVA;
Relocations = &SectionInfo->FileInfo.Relocations;
}
// If we don't have symbols available then we become a bit optimistic about multiblock ranges
if (!SymbolAvailable) {
// If we don't have a symbol available then assume all branches are valid for multiblock
SymbolMaxAddress = SectionMaxAddress;
SymbolMinAddress = EntryPoint;
}
DecodedMinAddress = EntryPoint;
@@ -1483,11 +1442,7 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
EraseBlock = true;
} else {
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" :
BlockIt->BlockStatus == DecodedBlockStatus::NOEXEC_INST ? "NoExec" :
BlockIt->BlockStatus == DecodedBlockStatus::BAD_RELOCATION ? "BadRelocation" :
"PartialDecode",
OpAddress);
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" : "NoExec", OpAddress);
}
break;
}
+8 -17
View File
@@ -4,12 +4,9 @@
#include "Interface/Core/X86Tables/X86Tables.h"
#include "Interface/IR/IR.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CodeCache.h>
#include <FEXCore/Utils/ThreadPoolAllocator.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/vector.h>
#include <FEXCore/fextl/robin_map.h>
#include <array>
#include <cstddef>
@@ -30,8 +27,6 @@ public:
SUCCESS,
INVALID_INST,
NOEXEC_INST,
PARTIAL_DECODE_INST,
BAD_RELOCATION,
};
// New Frontend decoding
@@ -54,7 +49,7 @@ public:
};
Decoder(FEXCore::Core::InternalThreadState* Thread);
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
const DecodedBlockInformation* GetDecodedBlockInfo() const {
return &BlockInfo;
@@ -63,6 +58,9 @@ public:
uint64_t DecodedMinAddress {};
uint64_t DecodedMaxAddress {~0ULL};
void SetSectionMaxAddress(uint64_t v) {
SectionMaxAddress = v;
}
void SetExternalBranches(fextl::set<uint64_t>* v) {
ExternalBranches = v;
}
@@ -88,8 +86,6 @@ private:
FEXCore::Context::ContextImpl* CTX;
const FEXCore::HLE::SyscallOSABI OSABI {};
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
bool DecodeInstructionImpl(uint64_t PC);
DecodedBlockStatus DecodeInstruction(uint64_t PC);
@@ -103,8 +99,7 @@ private:
uint8_t ReadByte();
std::optional<uint8_t> PeekByte(uint8_t Offset);
std::pair<uint64_t, bool> ReadData(uint8_t Size);
uint64_t ReadData(uint8_t Size);
void SkipBytes(uint8_t Size) {
InstructionSize += Size;
}
@@ -112,8 +107,6 @@ private:
bool NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
void DecodeREXIfValid(int8_t ExpectedOffset = -1);
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
Utils::PoolBufferWithTimedRetirement<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
@@ -123,7 +116,6 @@ private:
uint64_t ExecutableRangeEnd {};
bool ExecutableRangeWritable {};
bool HitNonExecutableRange {};
bool HitBadRelocation {};
const uint8_t* InstStream {};
IR::OpSize GetGPROpSize() const {
@@ -133,15 +125,16 @@ private:
static constexpr size_t MAX_INST_SIZE = 15;
uint8_t InstructionSize {};
std::array<uint8_t, MAX_INST_SIZE> Instruction;
uint8_t LastEscapePrefix {};
FEXCore::X86Tables::DecodedInst* DecodeInst;
// This is for multiblock data tracking
bool SymbolAvailable {false};
uint64_t EntryPoint {};
uint64_t MaxCondBranchForward {};
uint64_t MaxCondBranchBackwards {~0ULL};
uint64_t SymbolMaxAddress {};
uint64_t SymbolMinAddress {~0ULL};
uint64_t SectionMaxAddress {~0ULL};
uint64_t SectionMinAddress {};
uint64_t NextBlockStartAddress {~0ULL};
DecodedBlockInformation BlockInfo;
@@ -150,8 +143,6 @@ private:
fextl::set<uint64_t> VisitedBlocks;
fextl::set<uint64_t>* ExternalBranches {nullptr};
const fextl::robin_map<uint32_t, GuestRelocationType>* Relocations {nullptr};
// ModRM rm decoding
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
@@ -87,10 +87,12 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SINCOS>::handle)};
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
// Double Precision Binary
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle)};
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
@@ -218,21 +220,21 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
return true; \
}
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
return true; \
}
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64x2_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
return true; \
}
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
return true; \
}
// Unary
@@ -2,14 +2,14 @@
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
#include "Interface/IR/IR.h"
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
#include <arm_neon.h>
#endif
#include <cstring>
namespace FEXCore::CPU {
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
const auto is_using_words = (control & 1) != 0;
+45 -70
View File
@@ -43,28 +43,21 @@ DEF_BINOP_WITH_CONSTANT(Ror, rorv, ror)
DEF_OP(Constant) {
auto Op = IROp->C<IR::IROp_Constant>();
auto Dst = GetReg(Node);
const auto PadType = [Pad = Op->Pad]() {
switch (Pad) {
case IR::ConstPad::NoPad: return CPU::Arm64Emitter::PadType::NOPAD;
case IR::ConstPad::DoPad: return CPU::Arm64Emitter::PadType::DOPAD;
default: return CPU::Arm64Emitter::PadType::AUTOPAD;
}
}();
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Op->Constant, PadType, Op->MaxBytes);
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Op->Constant);
}
DEF_OP(EntrypointOffset) {
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
auto Constant = Entry + Op->Offset;
auto Dst = GetReg(Node);
uint64_t Mask = ~0ULL;
const auto OpSize = IROp->Size;
if (OpSize == IR::OpSize::i32Bit) {
Mask = 0xFFFF'FFFFULL;
}
InsertGuestRIPMove(GetReg(Node), Constant & Mask);
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Constant & Mask);
}
DEF_OP(InlineConstant) {
@@ -379,7 +372,7 @@ DEF_OP(CondSubNZCV) {
DEF_OP(Neg) {
auto Op = IROp->C<IR::IROp_Neg>();
if (Op->Cond == IR::CondClass::AL) {
if (Op->Cond == FEXCore::IR::COND_AL) {
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src));
} else {
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src), MapCC(Op->Cond));
@@ -595,7 +588,7 @@ DEF_OP(ShiftFlags) {
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, TMP1, &Done);
cbz(EmitSize, TMP1, &Done);
{
// PF/SF/ZF/OF
if (OpSize >= IR::OpSize::i32Bit) {
@@ -659,7 +652,7 @@ DEF_OP(ShiftFlags) {
msr(ARMEmitter::SystemRegister::NZCV, TMP2);
}
}
(void)Bind(&Done);
Bind(&Done);
// TODO: Make RA less dumb so this can't happen (e.g. with late-kill).
if (PFOutput != PFTemp) {
@@ -676,7 +669,7 @@ DEF_OP(RotateFlags) {
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, Shift, &Done);
cbz(EmitSize, Shift, &Done);
{
// Extract the last bit shifted in to CF
const auto BitSize = IR::OpSizeToSize(Op->Size) * 8;
@@ -708,7 +701,7 @@ DEF_OP(RotateFlags) {
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
}
}
(void)Bind(&Done);
Bind(&Done);
}
DEF_OP(Extr) {
@@ -774,14 +767,14 @@ DEF_OP(PDep) {
// Now, they're copied, so we can start setting Dest (even if it overlaps with
// one of them). Handle early exit case
mov(EmitSize, Dest, 0);
(void)cbz(EmitSize, OrigMask, &Done);
cbz(EmitSize, OrigMask, &Done);
// Setup for first iteration
neg(EmitSize, T0, Mask);
and_(EmitSize, T0, T0, Mask);
// Main loop
(void)Bind(&NextBit);
Bind(&NextBit);
sbfx(EmitSize, T1, Input, 0, 1);
eor(EmitSize, Mask, Mask, T0);
and_(EmitSize, T0, T1, T0);
@@ -789,10 +782,10 @@ DEF_OP(PDep) {
orr(EmitSize, Dest, Dest, T0);
lsr(EmitSize, Input, Input, 1);
and_(EmitSize, T0, Mask, T1);
(void)cbnz(EmitSize, T0, &NextBit);
cbnz(EmitSize, T0, &NextBit);
// All done with nothing to do.
(void)Bind(&Done);
Bind(&Done);
}
}
@@ -828,27 +821,27 @@ DEF_OP(PExt) {
ARMEmitter::BackwardLabel NextBit;
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, Mask, &EarlyExit);
cbz(EmitSize, Mask, &EarlyExit);
mov(EmitSize, MaskReg, Mask);
mov(EmitSize, ValueReg, Input);
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
// Main loop
(void)Bind(&NextBit);
(void)cbz(EmitSize, MaskReg, &Done);
Bind(&NextBit);
cbz(EmitSize, MaskReg, &Done);
clz(EmitSize, BitReg, MaskReg);
lslv(EmitSize, ValueReg, ValueReg, BitReg);
lslv(EmitSize, MaskReg, MaskReg, BitReg);
extr(EmitSize, Dest, Dest, ValueReg, OpSizeBitsM1);
bfc(EmitSize, MaskReg, OpSizeBitsM1, 1);
(void)b(&NextBit);
b(&NextBit);
// Early exit
(void)Bind(&EarlyExit);
Bind(&EarlyExit);
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
// All done with nothing to do.
(void)Bind(&Done);
Bind(&Done);
}
}
@@ -916,7 +909,7 @@ DEF_OP(Div) {
eor(EmitSize, TMP1, TMP1, Upper);
// If the sign bit matches then the result is zero
(void)cbz(EmitSize, TMP1, &Only64Bit);
cbz(EmitSize, TMP1, &Only64Bit);
// Long divide
{
@@ -924,7 +917,7 @@ DEF_OP(Div) {
mov(EmitSize, TMP2, Lower);
mov(EmitSize, TMP3, Divisor);
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.LDIVHandler));
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP4);
@@ -935,17 +928,17 @@ DEF_OP(Div) {
mov(EmitSize, Remainder, TMP2);
// Skip 64-bit path
(void)b(&LongDIVRet);
b(&LongDIVRet);
}
(void)Bind(&Only64Bit);
Bind(&Only64Bit);
// 64-Bit only
{
sdiv(EmitSize, Quotient, Lower, Divisor);
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
}
(void)Bind(&LongDIVRet);
Bind(&LongDIVRet);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown DIV Size: {}", OpSize); break;
@@ -999,7 +992,7 @@ DEF_OP(UDiv) {
// Check the upper bits for zero
// If the upper bits are zero then we can do a 64-bit divide
(void)cbz(EmitSize, Upper, &Only64Bit);
cbz(EmitSize, Upper, &Only64Bit);
// Long divide
{
@@ -1007,7 +1000,7 @@ DEF_OP(UDiv) {
mov(EmitSize, TMP2, Lower);
mov(EmitSize, TMP3, Divisor);
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.LUDIVHandler));
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP4);
@@ -1018,17 +1011,17 @@ DEF_OP(UDiv) {
mov(EmitSize, Remainder, TMP2);
// Skip 64-bit path
(void)b(&LongDIVRet);
b(&LongDIVRet);
}
(void)Bind(&Only64Bit);
Bind(&Only64Bit);
// 64-Bit only
{
udiv(EmitSize, Quotient, Lower, Divisor);
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
}
(void)Bind(&LongDIVRet);
Bind(&LongDIVRet);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUDIV Size: {}", OpSize); break;
@@ -1053,19 +1046,24 @@ DEF_OP(Popcount) {
if (CTX->HostFeatures.SupportsCSSC) {
switch (OpSize) {
case IR::OpSize::i8Bit:
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i16Bit:
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i32Bit: cnt(ARMEmitter::Size::i32Bit, Dst, Src); break;
case IR::OpSize::i64Bit: cnt(ARMEmitter::Size::i64Bit, Dst, Src); break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
case IR::OpSize::i8Bit:
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i16Bit:
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i32Bit:
cnt(ARMEmitter::Size::i32Bit, Dst, Src);
break;
case IR::OpSize::i64Bit:
cnt(ARMEmitter::Size::i64Bit, Dst, Src);
break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
}
} else {
}
else {
switch (OpSize) {
case IR::OpSize::i8Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
@@ -1197,19 +1195,6 @@ DEF_OP(Rev) {
}
}
DEF_OP(Rbit) {
auto Op = IROp->C<IR::IROp_Rbit>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = ConvertSize48(IROp);
const auto Dst = GetReg(Node);
const auto Src = GetReg(Op->Src);
rbit(EmitSize, Dst, Src);
}
DEF_OP(Bfi) {
auto Op = IROp->C<IR::IROp_Bfi>();
const auto EmitSize = ConvertSize(IROp);
@@ -1293,16 +1278,6 @@ DEF_OP(Sbfe) {
sbfx(ConvertSize(IROp), Dst, Src, Op->lsb, Op->Width);
}
DEF_OP(MaskGenerateFromBitWidth) {
auto Op = IROp->C<IR::IROp_MaskGenerateFromBitWidth>();
auto BitWidth = GetReg(Op->BitWidth);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1);
cmp(ARMEmitter::Size::i64Bit, BitWidth, 0);
lslv(ARMEmitter::Size::i64Bit, TMP2, TMP1, BitWidth);
csinv(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1, TMP2, ARMEmitter::Condition::CC_EQ);
}
DEF_OP(Select) {
auto Op = IROp->C<IR::IROp_Select>();
const auto OpSize = IROp->Size;
@@ -11,32 +11,36 @@ $end_info$
#include <FEXCore/Core/Thunks.h>
namespace FEXCore::CPU {
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl& CTX, FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
switch (Op) {
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return CTX.Dispatcher->GetExitFunctionLinkerAddress();
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
break;
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op)); break;
}
return ~0ULL;
}
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum) {
Relocation MoveABI {};
MoveABI.NamedThunkMove.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE};
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
MoveABI.NamedThunkMove.Offset = CurrentCursor - CodeData.BlockBegin;
MoveABI.NamedThunkMove.Symbol = Sum;
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
// Pointers are required to fit within 48-bit VA space.
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, FEXCore::CPU::Arm64Emitter::PadType::AUTOPAD, 6);
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, false);
Relocations.emplace_back(MoveABI);
}
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
uint64_t Pointer = GetNamedSymbolLiteral(*CTX, Op);
uint64_t Pointer = GetNamedSymbolLiteral(Op);
NamedSymbolLiteralPair Lit {
Arm64JITCore::NamedSymbolLiteralPair Lit {
.Lit = Pointer,
.MoveABI =
{
@@ -44,77 +48,87 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
{
.Header =
{
.Offset = 0, // Set by PlaceNamedSymbolLiteral
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
},
.Symbol = Op,
.Offset = 0,
},
},
};
return Lit;
}
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit) {
switch (Lit.MoveABI.Header.Type) {
case RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL:
case RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
Lit.MoveABI.Header.Offset = GetCursorOffset();
break;
}
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
default: ERROR_AND_DIE_FMT("Unknown relocation type for {}", __FUNCTION__);
}
BindOrRestart(&Lit.Loc);
Bind(&Lit.Loc);
dc64(Lit.Lit);
Relocations.emplace_back(Lit.MoveABI);
}
auto Arm64JITCore::InsertGuestRIPLiteral(uint64_t GuestRIP) -> NamedSymbolLiteralPair {
return {
.Lit = GuestRIP,
.MoveABI =
{
.GuestRIP = {.Header =
{
.Offset = 0, // Set by PlaceNamedSymbolLiteral
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL,
},
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
.GuestRIP = GuestRIP},
},
};
}
void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant) {
Relocation MoveABI {};
MoveABI.GuestRIP.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE};
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
MoveABI.GuestRIP.GuestRIP = Constant;
MoveABI.GuestRIP.RegisterIndex = Reg.Idx();
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
MoveABI.GuestRIPMove.Offset = CurrentCursor - CodeData.BlockBegin;
MoveABI.GuestRIPMove.GuestRIP = Constant;
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
// Pointers are required to fit within 48-bit VA space.
// TODO: Force 6-byte `MaxSize`, with sign extension to 64-bit. Current code not smart enough to handle negatives.
// 48-bit sign extension works because x86-64 guests only receive 47-bit VA space, with 48-bit being reserved for kernel.
// Additional quirk, "canonical" 48-bit pointers on x86-64, sign extend the 48-bit as well (Which is why kernel pointers are negative).
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, FEXCore::CPU::Arm64Emitter::PadType::AUTOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, false);
Relocations.emplace_back(MoveABI);
}
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations(uint64_t GuestBaseAddress) {
// Rebase relocations to library base address
for (auto& Relocation : Relocations) {
switch (Relocation.Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE:
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
Relocation.GuestRIP.GuestRIP -= GuestBaseAddress;
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
const char* EntryRelocations) {
size_t DataIndex {};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
dc64(Pointer);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
// XXX: Reenable once the JIT Object Cache is upstream
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
default:;
}
}
return std::move(Relocations);
return true;
}
} // namespace FEXCore::CPU
+53 -33
View File
@@ -62,27 +62,27 @@ DEF_OP(CASPair) {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
(void)Bind(&LoopTop);
Bind(&LoopTop);
// This instruction sequence must be synced with HandleCASPAL_Armv8.
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
cmp(EmitSize, TMP2, Expected0);
ccmp(EmitSize, TMP3, Expected1, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
stlxp(EmitSize, TMP2, Desired0, Desired1, MemSrc);
(void)cbnz(EmitSize, TMP2, &LoopTop);
cbnz(EmitSize, TMP2, &LoopTop);
mov(EmitSize, Dst0, Expected0);
mov(EmitSize, Dst1, Expected1);
(void)b(&LoopExpected);
b(&LoopExpected);
(void)Bind(&LoopNotExpected);
Bind(&LoopNotExpected);
mov(EmitSize, Dst0, TMP2.R());
mov(EmitSize, Dst1, TMP3.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
(void)Bind(&LoopExpected);
Bind(&LoopExpected);
// Restore
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
@@ -114,7 +114,7 @@ DEF_OP(CAS) {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
if (IROp->Size == IR::OpSize::i8Bit) {
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
@@ -123,18 +123,38 @@ DEF_OP(CAS) {
} else {
cmp(EmitSize, TMP2, Expected);
}
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
stlxr(SubEmitSize, TMP3, Desired, MemSrc);
(void)cbnz(EmitSize, TMP3, &LoopTop);
cbnz(EmitSize, TMP3, &LoopTop);
mov(EmitSize, Dst, Expected);
(void)b(&LoopExpected);
b(&LoopExpected);
(void)Bind(&LoopNotExpected);
Bind(&LoopNotExpected);
mov(EmitSize, Dst, TMP2.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
(void)Bind(&LoopExpected);
Bind(&LoopExpected);
}
}
DEF_OP(AtomicXor) {
auto Op = IROp->C<IR::IROp_AtomicXor>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
auto MemSrc = GetReg(Op->Addr);
auto Src = GetReg(Op->Value);
if (CTX->HostFeatures.SupportsAtomics) {
steorl(SubEmitSize, Src, MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
eor(EmitSize, TMP2, TMP2, Src);
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
cbnz(EmitSize, TMP2, &LoopTop);
}
}
@@ -159,10 +179,10 @@ DEF_OP(AtomicSwap) {
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
stlxr(SubEmitSize, TMP4, Src, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
ubfm(EmitSize, GetReg(Node), TMP2, 0, IR::OpSizeAsBits(OpSize) - 1);
}
}
@@ -179,11 +199,11 @@ DEF_OP(AtomicFetchAdd) {
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
add(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -201,11 +221,11 @@ DEF_OP(AtomicFetchSub) {
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
sub(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -223,11 +243,11 @@ DEF_OP(AtomicFetchAnd) {
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
and_(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -244,11 +264,11 @@ DEF_OP(AtomicFetchCLR) {
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
bic(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -265,11 +285,11 @@ DEF_OP(AtomicFetchOr) {
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
orr(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -286,11 +306,11 @@ DEF_OP(AtomicFetchXor) {
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
eor(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -306,20 +326,20 @@ DEF_OP(AtomicFetchNeg) {
// Use a CAS loop to avoid needing to emulate unaligned LLSC atomics
ldr(SubEmitSize, TMP2, MemSrc);
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
mov(EmitSize, TMP4, TMP2);
neg(EmitSize, TMP3, TMP2);
casal(SubEmitSize, TMP2, TMP3, MemSrc);
sub(EmitSize, TMP3, TMP2, TMP4);
(void)cbnz(EmitSize, TMP3, &LoopTop);
cbnz(EmitSize, TMP3, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
neg(EmitSize, TMP3, TMP2);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -329,7 +349,7 @@ DEF_OP(TelemetrySetValue) {
auto Op = IROp->C<IR::IROp_TelemetrySetValue>();
auto Src = GetReg(Op->Value);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.TelemetryValueAddresses[Op->TelemetryValueIndex]));
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.TelemetryValueAddresses[Op->TelemetryValueIndex]));
// Cortex fuses cmp+cset.
cmp(ARMEmitter::Size::i32Bit, Src, 0);
@@ -339,11 +359,11 @@ DEF_OP(TelemetrySetValue) {
stsetl(ARMEmitter::SubRegSize::i64Bit, TMP1, TMP2);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
}
#endif
}
+158 -55
View File
@@ -56,12 +56,12 @@ DEF_OP(ExitFunction) {
uint64_t NewRIP;
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionEC));
LoadConstant(ARMEmitter::Size::i64Bit, EC_CALL_CHECKER_PC_REG, NewRIP);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
} else {
#endif
@@ -141,7 +141,7 @@ DEF_OP(ExitFunction) {
if (!Op->CallReturnBlock.IsInvalid()) {
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
(void)adr(TMP1, &l_CallReturn);
adr(TMP1, &l_CallReturn);
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
} else {
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
@@ -149,17 +149,17 @@ DEF_OP(ExitFunction) {
} else if (Op->Hint == IR::BranchHint::CheckTF) {
ARMEmitter::ForwardLabel TFUnset;
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
InsertGuestRIPMove(TMP1, NewRIP);
cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, NewRIP);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
blr(TMP2);
(void)Bind(&TFUnset);
Bind(&TFUnset);
}
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
(void)Bind(&l_CallReturn);
#ifdef ARCHITECTURE_arm64ec
Bind(&l_CallReturn);
#ifdef _M_ARM_64EC
}
#endif
} else {
@@ -170,38 +170,40 @@ DEF_OP(ExitFunction) {
// First try to pop from the call-ret stack, otherwise follow the normal path (but ending in a ret)
ldp<ARMEmitter::IndexType::POST>(TMP1, TMP2, REG_CALLRET_SP, 0x10);
sub(TMP1, TMP1, RipReg.X());
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
}
// L1 Cache
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
// L1Mask is pre-shifted.
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
add(TMP1, TMP1, TMP2);
// arithmetic. ubfiz+add is marginally faster on Firestorm than
// and+add(shift). Same performance on Cortex.
static_assert(LookupCache::L1_ENTRIES_MASK == ((1u << 20) - 1));
ubfiz(ARMEmitter::Size::i64Bit, TMP4, RipReg, 4, 20);
add(TMP1, TMP1, TMP4);
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
// Note: sub+cbnz used over cmp+br to preserve flags.
sub(TMP1, TMP1, RipReg.X());
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
(void)Bind(&SkipFullLookup);
Bind(&SkipFullLookup);
if (Op->Hint == IR::BranchHint::Call) {
ARMEmitter::ForwardLabel l_CallReturn;
if (!Op->CallReturnBlock.IsInvalid()) {
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
(void)adr(TMP1, &l_CallReturn);
adr(TMP1, &l_CallReturn);
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
} else {
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
}
blr(TMP2);
(void)Bind(&l_CallReturn);
Bind(&l_CallReturn);
} else if (Op->Hint == IR::BranchHint::Return) {
ret(TMP2);
} else {
@@ -222,7 +224,7 @@ DEF_OP(CondJump) {
auto TrueTargetLabel = JumpTarget(Op->TrueBlock);
if (Op->FromNZCV) {
b_OrRestart(MapCC(Op->Cond), TrueTargetLabel);
b(MapCC(Op->Cond), TrueTargetLabel);
} else {
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
@@ -233,18 +235,18 @@ DEF_OP(CondJump) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1), "CondJump: Expected GPR");
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
if (Op->Cond == IR::CondClass::EQ) {
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
cbz_OrRestart(Size, Reg, TrueTargetLabel);
} else if (Op->Cond == IR::CondClass::NEQ) {
cbz(Size, Reg, TrueTargetLabel);
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
cbnz_OrRestart(Size, Reg, TrueTargetLabel);
} else if (Op->Cond == IR::CondClass::TSTZ) {
cbnz(Size, Reg, TrueTargetLabel);
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
tbz_OrRestart(Reg, Const, TrueTargetLabel);
} else if (Op->Cond == IR::CondClass::TSTNZ) {
tbz(Reg, Const, TrueTargetLabel);
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
tbnz_OrRestart(Reg, Const, TrueTargetLabel);
tbnz(Reg, Const, TrueTargetLabel);
} else {
LOGMAN_THROW_A_FMT(false, "CondJump expected simple condition");
}
@@ -260,10 +262,16 @@ DEF_OP(Syscall) {
// X1: ThreadState
// X2: Pointer to SyscallArguments
FEXCore::IR::SyscallFlags Flags = Op->Flags;
PushDynamicRegs(TMP1);
uint32_t GPRSpillMask = ~0U;
uint32_t FPRSpillMask = ~0U;
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) == FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
// Need to spill all caller saved registers still
GPRSpillMask = CALLER_GPR_MASK;
FPRSpillMask = CALLER_FPR_MASK;
}
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
@@ -283,8 +291,8 @@ DEF_OP(Syscall) {
str(GetReg(Op->Header.Args[i]).X(), ARMEmitter::Reg::rsp, i * 8);
}
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerObj));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerFunc));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc));
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
// SP supporting move
@@ -297,22 +305,117 @@ DEF_OP(Syscall) {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
// Result is now in x0
// Fix the stack and any values that were stepped on
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
// Result is now in x0
// Fix the stack and any values that were stepped on
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
// Now the registers we've spilled are back in their original host registers
// We can safely claim we are no longer in a syscall
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
// Now the registers we've spilled are back in their original host registers
// We can safely claim we are no longer in a syscall
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
PopDynamicRegs();
PopDynamicRegs();
const auto OSABI = CTX->SyscallHandler->GetOSABI();
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
// Move result to its destination register.
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
}
}
}
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
// Move result to its destination register.
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
DEF_OP(InlineSyscall) {
auto Op = IROp->C<IR::IROp_InlineSyscall>();
// Arguments are passed as follows:
// X8: SyscallNumber - RA INTERSECT
// X0: Arg0 & Return
// X1: Arg1
// X2: Arg2
// X3: Arg3
// X4: Arg4 - RA INTERSECT
// X5: Arg5 - RA INTERSECT
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS - 1> RegArgs = {
{ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5}};
bool Intersects {};
// We always need to spill x8 since we can't know if it is live at this SSA location
uint32_t SpillMask = 1U << 8;
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
break;
}
auto Reg = GetReg(Op->Header.Args[i]);
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
SpillMask |= (1U << Reg.Idx());
Intersects = true;
}
}
// Ordering is incredibly important here
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
// Only spill the registers that intersect with our usage
SpillStaticRegs(TMP1, false, SpillMask);
// Now that we are spilled, store in the state that we are in a syscall
// Still without overwriting registers that matter
// 16bit LoadConstant to be a single instruction
// We must always spill at least one register (x8) so this value always has a bit set
// This gives the signal handler a value to check to see if we are in a syscall at all
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
// Now that we have claimed to be a syscall we can set up the arguments
const auto EmitSize = CTX->Config.Is64BitMode() ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto EmitSubSize = CTX->Config.Is64BitMode() ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
if (Intersects) {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
break;
}
auto Reg = GetReg(Op->Header.Args[i]);
if (SpillMask & (1U << Reg.Idx())) {
// In the case of intersection with x4, x5, or x8 then these are currently SRA
// for registers RAX, RDX, and RSP. Which have just been spilled
// Just load back from the context.
auto Correlation = GetX86RegRelationToARMReg(Reg);
LOGMAN_THROW_A_FMT(Correlation != X86State::REG_INVALID, "Invalid register mapping");
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[Correlation]));
} else {
mov(EmitSize, RegArgs[i].R(), Reg);
}
}
} else {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
break;
}
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i]));
}
}
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, Op->HostSyscallNumber);
svc(0);
// On updated signal mask we can receive a signal RIGHT HERE
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
// Now that we are done in the syscall we need to carefully peel back the state
// First unspill the registers from before
FillStaticRegs(false, SpillMask, ~0U, ARMEmitter::Reg::r8, ARMEmitter::Reg::r1);
// Now the registers we've spilled are back in their original host registers
// We can safely claim we are no longer in a syscall
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
// Result is now in x0
// Move result to its destination register
mov(EmitSize, GetReg(Node), ARMEmitter::Reg::r0);
}
}
@@ -328,7 +431,8 @@ DEF_OP(Thunk) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr));
InsertNamedThunkRelocation(ARMEmitter::Reg::r2, Op->ThunkNameHash);
auto thunkFn = static_cast<Context::ContextImpl*>(ThreadState->CTX)->ThunkHandler->LookupThunk(Op->ThunkNameHash);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
@@ -354,7 +458,7 @@ DEF_OP(ValidateCode) {
while (len >= Size) {
LoadData();
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &Fail);
len -= Size;
Offset += Size;
}
@@ -382,10 +486,10 @@ DEF_OP(ValidateCode) {
ARMEmitter::ForwardLabel End;
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
b_OrRestart(&End);
BindOrRestart(&Fail);
b(&End);
Bind(&Fail);
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
BindOrRestart(&End);
Bind(&End);
}
DEF_OP(ThreadRemoveCodeEntry) {
@@ -397,10 +501,9 @@ DEF_OP(ThreadRemoveCodeEntry) {
// X1: RIP
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
// TODO: Relocations don't seem to be wired up to this...?
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry, CPU::Arm64Emitter::PadType::AUTOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadRemoveCodeEntryFromJIT));
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
@@ -425,8 +528,8 @@ DEF_OP(CPUID) {
// x0 = CPUID Handler
// x1 = CPUID Function
// x2 = CPUID Leaf
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.CPUIDObj));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.CPUIDFunction));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction));
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, TMP2);
@@ -466,8 +569,8 @@ DEF_OP(XGetBV) {
// x0 = CPUID Handler
// x1 = XCR Function
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.CPUIDObj));
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.XCRFunction));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
} else {
@@ -423,11 +423,11 @@ DEF_OP(Vector_FToI) {
const auto Mask = PRED_TMP_32B.Merging();
switch (Op->Round) {
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
}
} else {
const auto IsScalar = ElementSize == OpSize;
@@ -449,21 +449,21 @@ DEF_OP(Vector_FToI) {
}
switch (Op->Round) {
case IR::RoundMode::Nearest: ROUNDING_FN(frintn); break;
case IR::RoundMode::NegInfinity: ROUNDING_FN(frintm); break;
case IR::RoundMode::PosInfinity: ROUNDING_FN(frintp); break;
case IR::RoundMode::TowardsZero: ROUNDING_FN(frintz); break;
case IR::RoundMode::Host: ROUNDING_FN(frinti); break;
case IR::Round_Nearest.Val: ROUNDING_FN(frintn); break;
case IR::Round_Negative_Infinity.Val: ROUNDING_FN(frintm); break;
case IR::Round_Positive_Infinity.Val: ROUNDING_FN(frintp); break;
case IR::Round_Towards_Zero.Val: ROUNDING_FN(frintz); break;
case IR::Round_Host.Val: ROUNDING_FN(frinti); break;
}
#undef ROUNDING_FN
} else {
switch (Op->Round) {
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
}
}
}
@@ -539,11 +539,11 @@ DEF_OP(Vector_F64ToI32) {
// Then convert to integers using fcvtzs.
auto CVTReg = Dst.Z();
switch (Round) {
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::TowardsZero: CVTReg = Vector.Z(); break;
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::Round_Towards_Zero.Val: CVTReg = Vector.Z(); break;
case IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
}
fcvtzs(Dst.Z(), ARMEmitter::SubRegSize::i32Bit, Mask, CVTReg, ARMEmitter::SubRegSize::i64Bit);
@@ -567,11 +567,11 @@ DEF_OP(Vector_F64ToI32) {
///< Round float to integral depending on rounding mode.
switch (Round) {
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::TowardsZero: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
}
// Now narrow from f64 to f32.
@@ -1,35 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/fextl/vector.h>
#include <cstdint>
namespace FEXCore::CPU {
union Relocation;
} // namespace FEXCore::CPU
namespace FEXCore::Core {
struct DebugDataSubblock {
uint32_t HostCodeOffset;
uint32_t HostCodeSize;
};
struct DebugDataGuestOpcode {
uint64_t GuestEntryOffset;
ptrdiff_t HostEntryOffset;
};
/**
* @brief Contains debug data for a block of code for later debugger analysis
*
* Needs to remain around for as long as the code could be executed at least
*/
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
fextl::vector<DebugDataSubblock> Subblocks;
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
};
} // namespace FEXCore::Core
+180 -206
View File
@@ -1,7 +1,7 @@
// SPDX-License-Identifier: MIT
/*
$info$
glossary: Splatter ~ a code generator backend that concatenates configurable macros instead of doing isel
glossary: Splatter ~ a code generator backend that concaternates configurable macros instead of doing isel
glossary: IR ~ Intermediate Representation, our high-level opcode representation, loosely modeling arm64
glossary: SSA ~ Single Static Assignment, a form of representing IR in memory
glossary: Basic Block ~ A block of instructions with no control flow, terminated by control flow
@@ -11,12 +11,15 @@ desc: Main glue logic of the arm64 splatter backend
$end_info$
*/
#include "Common/SoftFloat.h"
#include "FEXCore/Utils/Telemetry.h"
#include "FEXCore/Utils/TypeDefines.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/JIT/DebugData.h"
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Utils/MemberFunctionToPointer.h"
@@ -27,16 +30,15 @@ $end_info$
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/LongJump.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <cstdio>
#include <cstring>
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include <stdio.h>
#include <unistd.h>
#include <string.h>
#include <limits>
namespace {
struct DivRem {
@@ -133,8 +135,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
const auto Src1 = GetVReg(IROp->Args[0]);
fmov(VTMP1.S(), Src1.S());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -151,8 +153,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
const auto Src1 = GetVReg(IROp->Args[0]);
fmov(VTMP1.D(), Src1.D());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -176,8 +178,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
mov(ARMEmitter::Size::i32Bit, TMP2, Src1);
}
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -194,8 +196,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
const auto Src1 = GetVReg(IROp->Args[0]);
mov(VTMP1.Q(), Src1.Q());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -212,8 +214,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
const auto Src1 = GetVReg(IROp->Args[0]);
mov(VTMP1.Q(), Src1.Q());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -230,8 +232,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
const auto Src1 = GetVReg(IROp->Args[0]);
fmov(VTMP1.D(), Src1.D());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -254,8 +256,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
fmov(VTMP1.D(), Src1.D());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -276,8 +278,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
fmov(VTMP1.D(), Src1.D());
fmov(VTMP2.D(), Src2.D());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -294,8 +296,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
const auto Src1 = GetVReg(IROp->Args[0]);
mov(VTMP1.Q(), Src1.Q());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -312,8 +314,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
const auto Src1 = GetVReg(IROp->Args[0]);
mov(VTMP1.Q(), Src1.Q());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -330,8 +332,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
const auto Src1 = GetVReg(IROp->Args[0]);
mov(VTMP1.Q(), Src1.Q());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -351,8 +353,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
mov(VTMP1.Q(), Src1.Q());
mov(VTMP2.Q(), Src2.Q());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -369,8 +371,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
const auto Src1 = GetVReg(IROp->Args[0]);
mov(VTMP1.Q(), Src1.Q());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -394,8 +396,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
mov(VTMP1.Q(), Src1.Q());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -416,8 +418,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
mov(VTMP1.Q(), Src1.Q());
mov(VTMP2.Q(), Src2.Q());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -434,8 +436,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
// tmp2 (x1/x11): source 2
// tmp3 (x2/x12): source 3
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
stp<ARMEmitter::IndexType::PRE>(TMP1, ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
@@ -476,8 +478,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
mov(VTMP2.Q(), Src2.Q());
movz(ARMEmitter::Size::i32Bit, TMP1, Control);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
blr(TMP2);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
@@ -493,7 +495,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
}
}
static void DirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record, bool Call) {
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record, bool Call) {
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
uintptr_t CallerAddress = JumpThunkStartAddress + Record->CallerOffset;
auto BranchOffset = JumpThunkStartAddress / 4 - CallerAddress / 4;
@@ -511,12 +513,11 @@ static void DirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record,
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(CallerAddress), 4);
}
static void IndirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record) {
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
uint32_t BranchInst = 0;
ARMEmitter::Emitter BranchEmit(reinterpret_cast<uint8_t*>(&BranchInst), 4);
// Restore branch +2 instructions to jump to the linker block
BranchEmit.b(0x2);
BranchEmit.b(0x8);
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(BranchInst, std::memory_order::relaxed);
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
@@ -533,13 +534,12 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
if (TFSet) {
// If TF is set, the cache must be skipped as different code needs to be generated.
Frame->State.rip = GuestRip;
return Frame->Pointers.DispatcherLoopTop;
return Frame->Pointers.Common.DispatcherLoopTop;
} else {
{
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
auto lk_inval =
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
HostCode = Thread->LookupCache->FindBlock(Thread, GuestRip);
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
HostCode = Thread->LookupCache->FindBlock(GuestRip);
}
if (!HostCode) {
// Hold a reference to the code buffer, to avoid linking unmapped code if compilation triggers a recreation.
@@ -564,7 +564,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
// Lock here is necessary to prevent simultaneous linking and delinking
auto lk = Thread->LookupCache->AcquireWriteLock();
auto lk = Thread->LookupCache->AcquireLock();
// For non-calls, this would extend into the block's code, however that's fine as an out-of-range adr would never
// be generated avoiding any false positives.
@@ -577,12 +577,14 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
if (KnownCallMarkerInst == ExpectedKnownCallMarkerInst) {
BranchEmit.bl(BranchOffset);
Thread->LookupCache->AddBlockLink(
GuestRip, Record, [](FEXCore::Context::ExitFunctionLinkData* Record) { DirectBlockDelinker(Record, true); }, lk);
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
DirectBlockDelinker(Frame, Record, true);
});
} else {
BranchEmit.b(BranchOffset);
Thread->LookupCache->AddBlockLink(
GuestRip, Record, [](FEXCore::Context::ExitFunctionLinkData* Record) { DirectBlockDelinker(Record, false); }, lk);
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
DirectBlockDelinker(Frame, Record, false);
});
}
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(CallerAddress)).store(BranchInst, std::memory_order::relaxed);
@@ -590,7 +592,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
} else {
// This case is common between calls and jumps as the thunk callsite can be left untouched.
std::atomic_ref<uint64_t>(Record->HostCode).store(HostCode, std::memory_order::seq_cst);
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
// Make memory write visible to other threads reading the same location
asm volatile("dc cvau, %0; dsb ish" : : "r"(Record->HostCode) :);
#endif
@@ -601,7 +603,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(LdrInst, std::memory_order::relaxed);
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker, lk);
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker);
}
return HostCode;
@@ -622,46 +624,59 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
RAPass->AddRegisters(IR::RegClass::GPR, GeneralRegisters.size());
RAPass->AddRegisters(IR::RegClass::GPRFixed, StaticRegisters.size());
RAPass->AddRegisters(IR::RegClass::FPR, GeneralFPRegisters.size());
RAPass->AddRegisters(IR::RegClass::FPRFixed, StaticFPRegisters.size());
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
RAPass->PairRegs = PairRegisters;
{
// Set up pointers that the JIT needs to load
// Common
auto& Ptrs = ThreadState->CurrentFrame->Pointers;
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
Ptrs.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
Ptrs.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
Ptrs.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
Ptrs.MonoBackpatcherWrite = reinterpret_cast<uint64_t>(&Context::ContextImpl::MonoBackpatcherWrite);
Ptrs.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
Common.MonoBackpatcherWrite = reinterpret_cast<uint64_t>(&Context::ContextImpl::MonoBackpatcherWrite);
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
{
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
Ptrs.CPUIDFunction = PMF.GetConvertedPointer();
Common.CPUIDFunction = PMF.GetConvertedPointer();
}
{
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
Ptrs.XCRFunction = PMF.GetConvertedPointer();
Common.XCRFunction = PMF.GetConvertedPointer();
}
{
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::HLE::SyscallHandler::HandleSyscall);
Ptrs.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
Ptrs.SyscallHandlerFunc = PMF.GetVTableEntry(CTX->SyscallHandler);
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
Common.SyscallHandlerFunc = PMF.GetVTableEntry(CTX->SyscallHandler);
}
Ptrs.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore::ExitFunctionLink);
Ptrs.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
Ptrs.LDIV = reinterpret_cast<uint64_t>(LDIV);
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore::ExitFunctionLink);
// Platform Specific
auto& AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
}
CurrentCodeBuffer = CodeBuffers.GetLatest();
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
// Setup dynamic dispatch.
if (ParanoidTSO()) {
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
} else {
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
}
}
void Arm64JITCore::EmitDetectionString() {
@@ -673,13 +688,13 @@ void Arm64JITCore::EmitDetectionString() {
void Arm64JITCore::ClearCache() {
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
auto PrevCodeBuffer = CurrentCodeBuffer;
auto lk = PrevCodeBuffer->LookupCache->AcquireWriteLock();
std::lock_guard lk(PrevCodeBuffer->LookupCache->WriteLock);
auto CodeBuffer = GetEmptyCodeBuffer();
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
EmitDetectionString();
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache, lk);
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache);
}
Arm64JITCore::~Arm64JITCore() {}
@@ -725,48 +740,48 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
}
}
void Arm64JITCore::EmitTFCheck() {
ARMEmitter::ForwardLabel l_TFUnset;
ARMEmitter::ForwardLabel l_TFBlocked;
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
if (CheckTF) {
ARMEmitter::ForwardLabel l_TFUnset;
ARMEmitter::ForwardLabel l_TFBlocked;
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
(void)tbz(TMP1, 1, &l_TFBlocked);
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
tbz(TMP1, 1, &l_TFBlocked);
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
.FaultToTopAndGeneratedException = 1,
.Signal = Core::FAULT_SIGTRAP,
.TrapNo = X86State::X86_TRAPNO_DB,
.si_code = 2,
.err_code = 0,
};
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
.FaultToTopAndGeneratedException = 1,
.Signal = Core::FAULT_SIGTRAP,
.TrapNo = X86State::X86_TRAPNO_DB,
.si_code = 2,
.err_code = 0,
};
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGTRAP));
br(TMP1);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
(void)Bind(&l_TFBlocked);
// If TF was blocked for this instruction, unblock it for the next.
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)Bind(&l_TFUnset);
}
Bind(&l_TFBlocked);
// If TF was blocked for this instruction, unblock it for the next.
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
Bind(&l_TFUnset);
}
void Arm64JITCore::EmitSuspendInterruptCheck() {
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
// Trigger a fault if there are any pending interrupts
// Used only for suspend on WIN32 at the moment
@@ -774,26 +789,24 @@ void Arm64JITCore::EmitSuspendInterruptCheck() {
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
}
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
static constexpr uint16_t SuspendMagic {0xCAFE};
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
ARMEmitter::ForwardLabel l_NoSuspend;
(void)cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
brk(SuspendMagic);
(void)Bind(&l_NoSuspend);
Bind(&l_NoSuspend);
#endif
}
void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF) {
// Get the address of the JITCodeHeader and store in to the core state.
// Two instruction cost, each 1 cycle.
adr_OrRestart(TMP1, &HeaderLabel);
adr(TMP1, &HeaderLabel);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
if (CheckTF) {
EmitTFCheck();
}
EmitInterruptChecks(CheckTF);
if (SpillSlots) {
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
@@ -805,65 +818,36 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
}
}
EmitSuspendInterruptCheck();
}
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
const auto PrevNumAllocations = Relocations.size();
JumpTargets.clear();
CallReturnTargets.clear();
PendingJumpThunks.clear();
uint32_t SSACount = IR->GetSSACount();
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
this->Entry = Entry;
this->DebugData = DebugData;
this->IR = IR;
RequiresFarARM64Jumps = false;
SSANodeMultiplier = 24;
// Prepare restart via long jump in case branch encoding fails.
// This uses UncheckedLongJump since we don't implement std::longjmp in WoA setups
switch (static_cast<RestartOptions::Control>(FEXCore::UncheckedLongJump::SetJump(ThreadState->RestartJump))) {
case RestartOptions::Control::Incoming:
// Nothing
break;
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
case RestartOptions::Control::NeedsLargerJITSpace:
// Get rid of the claimed buffer immediately, we can't fit in it at all.
TempAllocator.UnclaimBuffer();
SSANodeMultiplier *= 2;
break;
default: LOGMAN_MSG_A_FMT("Unhandled Arm64 restart condition!");
}
uint32_t SSACount = IR->GetSSACount();
JumpTargets.clear();
CallReturnTargets.clear();
PendingJumpThunks.clear();
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
CodeData.EntryPoints.clear();
// Fairly excessive buffer range to make sure we don't overflow
// One page baseline, plus SSANodeMultipler bytes, plus another page for guard page.
const uint32_t DesiredBufferRange = AlignUp(FEXCore::Utils::FEX_PAGE_SIZE * 2 + SSACount * SSANodeMultiplier, FEXCore::Utils::FEX_PAGE_SIZE);
uint32_t BufferRange = 0x1000 + SSACount * 24;
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
// This minimizes lock contention of CodeBufferWriteMutex.
auto TempCodeBufferInfo = TempAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
auto TempCodeBuffer = TempCodeBufferInfo.Ptr;
const uint32_t UsableBufferRange = TempCodeBufferInfo.Size - FEXCore::Utils::FEX_PAGE_SIZE;
SetBuffer(TempCodeBuffer, UsableBufferRange);
ThreadState->JITGuardPage = reinterpret_cast<uintptr_t>(TempCodeBuffer) + UsableBufferRange;
ThreadState->JITGuardOverflowArgument = FEXCore::ToUnderlying(RestartOptions::Control::NeedsLargerJITSpace);
auto TempCodeBuffer = TempAllocator.ReownOrClaimBuffer(BufferRange);
SetBuffer(TempCodeBuffer, BufferRange);
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
// Put the code header at the start of the data block.
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
(void)Bind(&JITCodeHeaderLabel);
Bind(&JITCodeHeaderLabel);
JITCodeHeader* CodeHeader = GetCursorAddress<JITCodeHeader*>();
CursorIncrement(sizeof(JITCodeHeader));
@@ -908,10 +892,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// if there's a pending branch, and it is not fall-through
if (PendingTargetLabel && PendingTargetLabel != Target) {
if (PendingTargetLabel->Backward.Location) {
EmitSuspendInterruptCheck();
}
b_OrRestart(PendingTargetLabel);
b(PendingTargetLabel);
PendingTargetLabel = nullptr;
}
@@ -921,14 +902,14 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
const auto IsReturnTarget = CallReturnTargets.try_emplace(Node).first;
if (PendingTargetLabel) {
// If there is a fallthrough branch to this block, skip over the entrypoint code.
b_OrRestart(Target);
b(Target);
} else if (PendingCallReturnTargetLabel && PendingCallReturnTargetLabel != &IsReturnTarget->second) {
// If we just emitted a call, but the block we're now emitting is not the return block so don't fallthrough.
b_OrRestart(PendingCallReturnTargetLabel);
b(PendingCallReturnTargetLabel);
}
PendingCallReturnTargetLabel = nullptr;
BindOrRestart(&IsReturnTarget->second);
Bind(&IsReturnTarget->second);
CodeData.EntryPoints.emplace(BlockStartRIP, GetCursorAddress<uint8_t*>());
DebugData->GuestOpcodes.push_back({BlockIROp->GuestEntryOffset, GetCursorAddress<uint8_t*>() - CodeData.BlockBegin});
@@ -937,16 +918,18 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
if (PendingCallReturnTargetLabel) {
// If there is still a pending call return target, then the block we're emitting is not the return block so don't fallthrough.
b_OrRestart(PendingCallReturnTargetLabel);
b(PendingCallReturnTargetLabel);
PendingCallReturnTargetLabel = nullptr;
}
PendingTargetLabel = nullptr;
BindOrRestart(Target);
Bind(Target);
}
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
switch (IROp->Op) {
#define REGISTER_OP_RT(op, x) \
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, CodeNode); break
#define REGISTER_OP(op, x) \
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, CodeNode); break
@@ -964,10 +947,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Make sure last branch is generated. It certainly can't be eliminated here.
if (PendingTargetLabel) {
if (PendingTargetLabel->Backward.Location) {
EmitSuspendInterruptCheck();
}
b_OrRestart(PendingTargetLabel);
b(PendingTargetLabel);
}
PendingTargetLabel = nullptr;
@@ -978,37 +958,31 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
ARMEmitter::ForwardLabel l_DoLink;
uint64_t ThunkAddress = GetCursorAddress<uint64_t>();
BindOrRestart(&PendingJumpThunk.Label);
b_OrRestart(&l_DoLink);
Bind(&PendingJumpThunk.Label);
b(&l_DoLink);
br(TMP1);
BindOrRestart(&l_DoLink);
Bind(&l_DoLink);
ldr(TMP1, &l_ExitLink);
blr(TMP1);
// This is a ExitFunctionLinkData struct
BindOrRestart(&l_ExitLink);
dc64(0); // HostCode
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
Bind(&l_ExitLink);
dc64(0); // HostCode
dc64(PendingJumpThunk.GuestRIP); // GuestRIP
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
}
BindOrRestart(&l_ExitLink);
PlaceNamedSymbolLiteral(InsertNamedSymbolLiteral(RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER));
Bind(&l_ExitLink);
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
// CodeSize not including the header or tail data.
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeBegin;
// Add the JitCodeTail (written later)
// Add the JitCodeTail
Align(alignof(JITCodeTail));
const auto JITBlockTailLocation = GetCursorAddress<uint8_t*>();
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
JITCodeTail JITBlockTail {
.RIP = Entry,
.GuestSize = Size,
.SpinLockFutex = 0,
.SingleInst = SingleInst,
};
auto JITBlockTailLocation = GetCursorAddress<uint8_t*>();
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
CursorIncrement(sizeof(JITCodeTail));
// Entries that live after the JITCodeTail.
// These entries correlate JIT code regions with guest RIP regions.
@@ -1026,13 +1000,23 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// FEXCore::Utils::vl64 GuestRIPOffset;
// };
const auto JITRIPEntriesBegin = JITBlockTailLocation + sizeof(JITBlockTail);
auto JITRIPEntriesBegin = GetCursorAddress<uint8_t*>();
// Put the block's RIP entry in the tail.
// This will be used for RIP reconstruction in the future.
// TODO: This needs to be a data RIP relocation once code caching works.
// Current relocation code doesn't support this feature yet.
JITBlockTail->RIP = Entry;
JITBlockTail->GuestSize = Size;
JITBlockTail->SingleInst = SingleInst;
JITBlockTail->SpinLockFutex = 0;
auto JITRIPEntriesLocation = JITRIPEntriesBegin;
{
// Store the RIP entries.
JITBlockTail.NumberOfRIPEntries = DebugData->GuestOpcodes.size();
JITBlockTail.OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
uintptr_t CurrentRIPOffset = 0;
uint64_t CurrentPCOffset = 0;
@@ -1048,20 +1032,14 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
}
}
SetCursorOffset(JITRIPEntriesLocation - CodeData.BlockBegin);
CursorIncrement(JITRIPEntriesLocation - JITRIPEntriesBegin);
Align();
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
// Finalize and write block tail data
JITBlockTail.Size = CodeData.Size;
{
auto PrevCur = GetCursorOffset();
memcpy(JITBlockTailLocation, &JITBlockTail, sizeof(JITBlockTail));
SetCursorOffset(JITBlockTailLocation - CodeData.BlockBegin + offsetof(JITCodeTail, RIP));
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(JITBlockTail.RIP));
SetCursorOffset(PrevCur);
}
JITBlockTail->Size = CodeData.Size;
// Migrate the compile output from temporary storage to the actual CodeBuffer.
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
@@ -1070,6 +1048,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Query size of generated code
const auto TempSize = GetCursorOffset();
LOGMAN_THROW_A_FMT(TempSize <= BufferRange, "Exceeded bounds of temporary buffer ({:#x} vs {:#x})", TempSize, BufferRange);
// Bring CodeBuffer up to date
{
@@ -1077,8 +1056,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
"doesn't match up!\n");
if (auto Prev = CheckCodeBufferUpdate()) {
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
auto lk = ThreadState->LookupCache->AcquireWriteLock();
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
}
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
@@ -1101,10 +1079,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
}
CodeBegin += Delta;
for (std::size_t Idx = PrevNumAllocations; Idx != Relocations.size(); ++Idx) {
Relocations[Idx].Header.Offset += CodeBuffers.LatestOffset;
}
// Copy over CodeBuffer contents
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
SetCursorOffset(CodeBuffers.LatestOffset + TempSize);
+74 -281
View File
@@ -10,39 +10,31 @@ $end_info$
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/JIT/Relocations.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <FEXCore/Utils/LongJump.h>
#include <CodeEmitter/Emitter.h>
#include <array>
#include <cstdint>
#include <functional>
#include <optional>
#include <utility>
#include <variant>
namespace FEXCore::Core {
struct InternalThreadState;
}
namespace FEXCore::Context {
struct ExitFunctionLinkData;
}
namespace FEXCore::IR {
class RegisterAllocationPass;
}
namespace FEXCore::CPU {
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
@@ -61,27 +53,14 @@ public:
}
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
const bool HostSupportsSVE128 {};
const bool HostSupportsSVE256 {};
const bool HostSupportsAVX256 {};
const bool HostSupportsRPRES {};
const bool HostSupportsAFP {};
struct RestartOptions {
enum class Control : uint64_t {
Incoming = 0,
EnableFarARM64Jumps = 1,
NeedsLargerJITSpace = 2,
};
};
// FEXCore makes assumptions in the JIT about certain conditions being true.
// In the rare case when those assumptions are broken, FEX needs to safely restart the JIT.
RestartOptions RestartControl {};
bool RequiresFarARM64Jumps {};
// Default to 6 instructions per SSA node.
uint32_t SSANodeMultiplier {24};
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
ARMEmitter::BiDirectionalLabel* PendingCallReturnTargetLabel {};
FEXCore::Context::ContextImpl* CTX {};
@@ -111,13 +90,11 @@ private:
[[nodiscard]]
ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const {
const auto RegClass = Reg.AsRegClass();
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::GPRFixed || RegClass == IR::RegClass::GPR, "Unexpected Class: {}", Reg.Class);
if (RegClass == IR::RegClass::GPRFixed) {
if (Reg.Class == IR::GPRFixedClass.Val) {
return StaticRegisters[Reg.Reg];
} else if (RegClass == IR::RegClass::GPR) {
} else if (Reg.Class == IR::GPRClass.Val) {
return GeneralRegisters[Reg.Reg];
}
@@ -136,13 +113,11 @@ private:
[[nodiscard]]
ARMEmitter::VRegister GetVReg(IR::PhysicalRegister Reg) const {
const auto RegClass = Reg.AsRegClass();
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::FPRFixed || RegClass == IR::RegClass::FPR, "Unexpected Class: {}", Reg.Class);
if (RegClass == IR::RegClass::FPRFixed) {
if (Reg.Class == IR::FPRFixedClass.Val) {
return StaticFPRegisters[Reg.Reg];
} else if (RegClass == IR::RegClass::FPR) {
} else if (Reg.Class == IR::FPRClass.Val) {
return GeneralFPRegisters[Reg.Reg];
}
@@ -160,8 +135,8 @@ private:
}
[[nodiscard]]
static IR::RegClass GetRegClass(IR::Ref Node) {
return IR::PhysicalRegister(Node).AsRegClass();
FEXCore::IR::RegisterClassType GetRegClass(IR::Ref Node) const {
return FEXCore::IR::RegisterClassType {IR::PhysicalRegister(Node).Class};
}
[[nodiscard]]
@@ -178,7 +153,7 @@ private:
// Converts IR-base shift type to ARMEmitter shift type.
// Will be a no-op, only a type conversion since the two definitions match.
[[nodiscard]]
static ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) {
ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
@@ -186,23 +161,18 @@ private:
}
[[nodiscard]]
static ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
return Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
}
[[nodiscard]]
static ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
return ConvertSize(Op);
}
[[nodiscard]]
static ARMEmitter::Size ConvertSize(IR::OpSize Size) {
return Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
"Invalid size");
@@ -214,105 +184,105 @@ private:
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
return ConvertSubRegSize16(Op->ElementSize);
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
LOGMAN_THROW_A_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
return ConvertSubRegSize16(ElementSize);
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
return ConvertSubRegSize8(Op->ElementSize);
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
return ConvertSubRegSize8(Op);
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
return ConvertSubRegSize8(Op);
}
[[nodiscard]]
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
return ARMEmitter::ToVectorSizePair(ConvertSubRegSize16(Op));
}
[[nodiscard]]
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
return ConvertSubRegSizePair16(Op);
}
[[nodiscard]]
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
return ConvertSubRegSizePair8(Op);
}
[[nodiscard]]
static ARMEmitter::Condition MapCC(IR::CondClass Cond) {
switch (Cond) {
case IR::CondClass::EQ: return ARMEmitter::Condition::CC_EQ;
case IR::CondClass::NEQ: return ARMEmitter::Condition::CC_NE;
case IR::CondClass::SGE: return ARMEmitter::Condition::CC_GE;
case IR::CondClass::SLT: return ARMEmitter::Condition::CC_LT;
case IR::CondClass::SGT: return ARMEmitter::Condition::CC_GT;
case IR::CondClass::SLE: return ARMEmitter::Condition::CC_LE;
case IR::CondClass::UGE: return ARMEmitter::Condition::CC_CS;
case IR::CondClass::ULT: return ARMEmitter::Condition::CC_CC;
case IR::CondClass::UGT: return ARMEmitter::Condition::CC_HI;
case IR::CondClass::ULE: return ARMEmitter::Condition::CC_LS;
case IR::CondClass::FLU: return ARMEmitter::Condition::CC_LT;
case IR::CondClass::FGE: return ARMEmitter::Condition::CC_GE;
case IR::CondClass::FLEU: return ARMEmitter::Condition::CC_LE;
case IR::CondClass::FGT: return ARMEmitter::Condition::CC_GT;
case IR::CondClass::FU:
case IR::CondClass::VS: return ARMEmitter::Condition::CC_VS;
case IR::CondClass::FNU:
case IR::CondClass::VC: return ARMEmitter::Condition::CC_VC;
case IR::CondClass::MI: return ARMEmitter::Condition::CC_MI;
case IR::CondClass::PL: return ARMEmitter::Condition::CC_PL;
ARMEmitter::Condition MapCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_FU:
case FEXCore::IR::COND_VS: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FNU:
case FEXCore::IR::COND_VC: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
}
}
[[nodiscard]]
static bool IsFPR(IR::RegClass Class) {
return Class == IR::RegClass::FPR || Class == IR::RegClass::FPRFixed;
bool IsFPR(IR::RegisterClassType Class) const {
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
}
[[nodiscard]]
static bool IsGPR(IR::RegClass Class) {
return Class == IR::RegClass::GPR || Class == IR::RegClass::GPRFixed;
bool IsGPR(IR::RegisterClassType Class) const {
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
}
[[nodiscard]]
static bool IsGPR(IR::Ref Node) {
bool IsGPR(IR::Ref Node) {
return IsGPR(GetRegClass(Node));
}
[[nodiscard]]
static bool IsFPR(IR::Ref Node) {
bool IsFPR(IR::Ref Node) {
return IsFPR(GetRegClass(Node));
}
[[nodiscard]]
static bool IsGPR(IR::OrderedNodeWrapper Wrap) {
return IsGPR(IR::PhysicalRegister(Wrap).AsRegClass());
bool IsGPR(IR::OrderedNodeWrapper Wrap) {
return IsGPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
}
[[nodiscard]]
static bool IsFPR(IR::OrderedNodeWrapper Wrap) {
return IsFPR(IR::PhysicalRegister(Wrap).AsRegClass());
bool IsFPR(IR::OrderedNodeWrapper Wrap) {
return IsFPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
}
[[nodiscard]]
@@ -345,187 +315,14 @@ private:
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
auto& Thunk = PendingJumpThunks.back();
BindOrRestart(&Thunk.Label);
Bind(&Thunk.Label);
if (Call) {
bl_OrRestart(&Thunk.Label);
bl(&Thunk.Label);
} else {
b_OrRestart(&Thunk.Label);
b(&Thunk.Label);
}
}
// Restart helpers
template<ARMEmitter::IsLabel T>
void bl_OrRestart(T* Label) {
if (bl(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<ARMEmitter::IsLabel T>
void b_OrRestart(T* Label) {
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<ARMEmitter::IsLabel T>
void b_OrRestart(ARMEmitter::Condition Cond, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)b(InvertCondition(Cond), &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (b(Cond, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<ARMEmitter::IsLabel T>
void cbz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)cbnz(s, rt, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (cbz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<ARMEmitter::IsLabel T>
void cbnz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)cbz(s, rt, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (cbnz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<ARMEmitter::IsLabel T>
void tbz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)tbnz(rt, Bit, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (tbz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<ARMEmitter::IsLabel T>
void tbnz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)tbz(rt, Bit, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (tbnz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<ARMEmitter::IsLabel T>
void adr_OrRestart(ARMEmitter::Register rd, T* Label) {
if (RequiresFarARM64Jumps) {
if (LongAddressGen(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Unable to encode long ADR.");
}
return;
}
if (adr(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<ARMEmitter::IsLabel T>
void adrp_OrRestart(ARMEmitter::Register rd, T* Label) {
if (RequiresFarARM64Jumps) {
if (LongAddressGen(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Unable to encode long ADRP.");
}
return;
}
if (adrp(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<ARMEmitter::IsLabel T>
void BindOrRestart(T* Label) {
if (Bind(Label)) {
return;
}
if (RequiresFarARM64Jumps) {
// This should have been caught before this point.
ERROR_AND_DIE_FMT("Unhandled long bind");
return;
}
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
IR::RegisterAllocationPass* RAPass {};
@@ -536,6 +333,8 @@ private:
* @name Relocations
* @{ */
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief A literal pair relocation object for named symbol literals
*/
@@ -572,30 +371,17 @@ private:
*/
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief Inserts a relocation for a constant value relative to the guest entrypoint
*
* @param Reg - The GPR to move the guest RIP in to
* @param Constant - The guest RIP that will be relocated
*/
NamedSymbolLiteralPair InsertGuestRIPLiteral(uint64_t GuestRIP);
/**
* @brief Place the named symbol literal relocation in memory
*
* @param Lit - Which literal to place
*/
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit);
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit);
fextl::vector<FEXCore::CPU::Relocation> Relocations;
/**
* Returns any relocations generated since the last call to TakeRelocations.
*
* GuestBaseAddress must match the base virtual address to which the
* input x86 binary is mapped.
*/
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) override;
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
/** @} */
@@ -618,16 +404,23 @@ private:
void Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize);
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
void EmitTFCheck();
void EmitSuspendInterruptCheck();
void EmitInterruptChecks(bool CheckTF);
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
// Runtime selection;
// Load and store TSO memory style
OpType RT_LoadMemTSO;
OpType RT_StoreMemTSO;
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
// Dynamic Dispatcher supporting operations
DEF_OP(ParanoidLoadMemTSO);
DEF_OP(ParanoidStoreMemTSO);
///< Unhandled handler
DEF_OP(Unhandled);
+269 -121
View File
@@ -21,7 +21,7 @@ DEF_OP(LoadContext) {
const auto Op = IROp->C<IR::IROp_LoadContext>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
auto Dst = GetReg(Node);
switch (OpSize) {
@@ -52,7 +52,7 @@ DEF_OP(LoadContext) {
DEF_OP(LoadContextPair) {
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst1 = GetReg(Op->OutValue1);
const auto Dst2 = GetReg(Op->OutValue2);
@@ -78,7 +78,7 @@ DEF_OP(StoreContext) {
const auto Op = IROp->C<IR::IROp_StoreContext>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
auto Src = GetZeroableReg(Op->Value);
switch (OpSize) {
@@ -110,7 +110,7 @@ DEF_OP(StoreContextPair) {
const auto Op = IROp->C<IR::IROp_StoreContextPair>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
auto Src1 = GetZeroableReg(Op->Value1);
auto Src2 = GetZeroableReg(Op->Value2);
@@ -135,11 +135,11 @@ DEF_OP(StoreContextPair) {
DEF_OP(LoadRegister) {
const auto Op = IROp->C<IR::IROp_LoadRegister>();
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == IR::GPRClass) {
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
} else if (Op->Class == IR::RegClass::FPR) {
} else if (Op->Class == IR::FPRClass) {
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
@@ -175,13 +175,12 @@ DEF_OP(LoadAF) {
DEF_OP(StoreRegister) {
const auto Op = IROp->C<IR::IROp_StoreRegister>();
const auto Reg = IR::PhysicalRegister(Node);
const auto RegClass = Reg.AsRegClass();
auto Reg = IR::PhysicalRegister(Node);
if (RegClass == IR::RegClass::GPRFixed) {
if (Reg.Class == IR::GPRFixedClass) {
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
mov(ARMEmitter::Size::i64Bit, GetReg(Reg), GetReg(Op->Value));
} else if (RegClass == IR::RegClass::FPRFixed) {
} else if (Reg.Class == IR::FPRFixedClass) {
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
@@ -194,7 +193,7 @@ DEF_OP(StoreRegister) {
mov(guest.Q(), host.Q());
}
} else {
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", RegClass);
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Reg.Class);
}
}
@@ -226,7 +225,7 @@ DEF_OP(LoadContextIndexed) {
const auto Index = GetReg(Op->Index);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
switch (Op->Stride) {
case 1:
case 2:
@@ -289,7 +288,7 @@ DEF_OP(StoreContextIndexed) {
const auto Index = GetReg(Op->Index);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Value = GetReg(Op->Value);
switch (Op->Stride) {
@@ -349,31 +348,12 @@ DEF_OP(StoreContextIndexed) {
}
}
DEF_OP(FormContextAddress) {
const auto Op = IROp->C<IR::IROp_FormContextAddress>();
const auto Index = GetReg(Op->Index);
const auto Dst = GetReg(Node);
switch (Op->Stride) {
case 1:
case 2:
case 4:
case 8:
case 16:
case 32: {
add(ARMEmitter::Size::i64Bit, Dst, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled FormContextAddress stride: {}", Op->Stride); break;
}
}
DEF_OP(SpillRegister) {
const auto Op = IROp->C<IR::IROp_SpillRegister>();
const auto OpSize = IROp->Size;
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetReg(Op->Value);
switch (OpSize) {
case IR::OpSize::i8Bit: {
@@ -414,7 +394,7 @@ DEF_OP(SpillRegister) {
}
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
}
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
} else if (Op->Class == FEXCore::IR::FPRClass) {
const auto Src = GetVReg(Op->Value);
switch (OpSize) {
@@ -453,7 +433,7 @@ DEF_OP(SpillRegister) {
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
}
} else {
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class);
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
}
}
@@ -462,7 +442,7 @@ DEF_OP(FillRegister) {
const auto OpSize = IROp->Size;
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
switch (OpSize) {
case IR::OpSize::i8Bit: {
@@ -503,7 +483,7 @@ DEF_OP(FillRegister) {
}
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
}
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
} else if (Op->Class == FEXCore::IR::FPRClass) {
const auto Dst = GetVReg(Node);
switch (OpSize) {
@@ -542,7 +522,7 @@ DEF_OP(FillRegister) {
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
}
} else {
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class);
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
}
}
@@ -579,14 +559,14 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, Const);
} else {
auto RegOffset = GetReg(Offset);
switch (OffsetType) {
case IR::MemOffsetType::SXTX:
switch (OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val:
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
case IR::MemOffsetType::UXTW:
case IR::MEM_OFFSET_UXTW.Val:
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
case IR::MemOffsetType::SXTW:
case IR::MEM_OFFSET_SXTW.Val:
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType); break;
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
}
}
}
@@ -613,20 +593,20 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
} else {
auto RegOffset = GetReg(Offset);
switch (OffsetType) {
case IR::MemOffsetType::SXTX:
switch (OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val:
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
break;
case IR::MemOffsetType::UXTW:
case IR::MEM_OFFSET_UXTW.Val:
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
break;
case IR::MemOffsetType::SXTW:
case IR::MEM_OFFSET_SXTW.Val:
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
break;
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType); break;
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType.Val); break;
}
}
return Tmp;
@@ -677,7 +657,7 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
// Note that we do nothing with the offset type and offset scale,
// since SVE loads and stores don't have the ability to perform an
// optional extension or shift as part of their behavior.
LOGMAN_THROW_A_FMT(OffsetType == IR::MemOffsetType::SXTX, "Currently only the default offset type (SXTX) is supported.");
LOGMAN_THROW_A_FMT(OffsetType.Val == IR::MEM_OFFSET_SXTX.Val, "Currently only the default offset type (SXTX) is supported.");
const auto RegOffset = GetReg(Offset);
return ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
@@ -690,7 +670,7 @@ DEF_OP(LoadMem) {
const auto MemReg = GetReg(Op->Addr);
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
switch (OpSize) {
@@ -724,7 +704,7 @@ DEF_OP(LoadMemPair) {
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
const auto Addr = GetReg(Op->Addr);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst1 = GetReg(Op->OutValue1);
const auto Dst2 = GetReg(Op->OutValue2);
@@ -752,13 +732,13 @@ DEF_OP(LoadMemTSO) {
const auto MemReg = GetReg(Op->Addr);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
}
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
@@ -780,7 +760,7 @@ DEF_OP(LoadMemTSO) {
// Half-barrier once back-patched.
nop();
}
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == IR::RegClass::GPR) {
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
@@ -795,7 +775,7 @@ DEF_OP(LoadMemTSO) {
// Half-barrier once back-patched.
nop();
}
} else if (Op->Class == IR::RegClass::GPR) {
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
@@ -912,7 +892,7 @@ DEF_OP(VLoadVectorMasked) {
// If the sign bit is zero then skip the load
ARMEmitter::ForwardLabel Skip {};
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
// Do the gather load for this element into the destination
switch (IROp->ElementSize) {
case IR::OpSize::i8Bit: ld1<ARMEmitter::SubRegSize::i8Bit>(TempDst.Q(), i, TempMemReg); break;
@@ -923,7 +903,7 @@ DEF_OP(VLoadVectorMasked) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
}
(void)Bind(&Skip);
Bind(&Skip);
if ((i + 1) != NumElements) {
// Handle register rename to save a move.
@@ -1013,7 +993,7 @@ DEF_OP(VStoreVectorMasked) {
// If the sign bit is zero then skip the load
ARMEmitter::ForwardLabel Skip {};
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
// Do the gather load for this element into the destination
switch (IROp->ElementSize) {
case IR::OpSize::i8Bit: st1<ARMEmitter::SubRegSize::i8Bit>(RegData.Q(), i, TempMemReg); break;
@@ -1024,7 +1004,7 @@ DEF_OP(VStoreVectorMasked) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
}
(void)Bind(&Skip);
Bind(&Skip);
if ((i + 1) != NumElements) {
// Handle register rename to save a move.
@@ -1040,7 +1020,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
ARMEmitter::VRegister IncomingDst, std::optional<ARMEmitter::Register> BaseAddr,
ARMEmitter::VRegister VectorIndexLow, std::optional<ARMEmitter::VRegister> VectorIndexHigh,
ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize, size_t DataElementOffsetStart,
size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize) {
size_t IndexElementOffsetStart, uint8_t OffsetScale) {
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid element size");
const auto PerformSMove = [this](IR::OpSize ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
@@ -1102,7 +1082,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
PerformMove(ElementSize, WorkingReg, MaskReg, i);
// Skip if the mask's sign bit isn't set
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
// Extract Index Element
if ((IndexElement * IR::OpSizeToSize(VectorIndexSize)) >= 16) {
@@ -1116,17 +1096,17 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
// Calculate memory position for this gather load
if (BaseAddr.has_value()) {
if (VectorIndexSize == IR::OpSize::i32Bit) {
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
} else {
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
}
} else {
///< In this case we have no base address, All addresses come from the vector register itself
if (VectorIndexSize == IR::OpSize::i32Bit) {
// Sign extend and shift in to the 64-bit register
sbfiz(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
sbfiz(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
} else {
lsl(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
lsl(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
}
}
@@ -1140,7 +1120,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, ElementSize); FEX_UNREACHABLE;
}
(void)Bind(&Skip);
Bind(&Skip);
}
if (NeedsDestTmp) {
@@ -1184,8 +1164,7 @@ DEF_OP(VLoadVectorGatherMasked) {
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
const bool SupportsSVELoad = (HostSupportsSVE128 || HostSupportsSVE256) &&
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) &&
VectorIndexSize == IROp->ElementSize && Op->AddrSize == IR::OpSize::i64Bit;
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) && VectorIndexSize == IROp->ElementSize;
if (SupportsSVELoad) {
uint8_t SVEScale = FEXCore::ilog2(OffsetScale);
@@ -1243,7 +1222,7 @@ DEF_OP(VLoadVectorGatherMasked) {
} else {
LOGMAN_THROW_A_FMT(!Is256Bit, "Can't emulate this gather load in the backend! Programming error!");
Emulate128BitGather(IROp->Size, IROp->ElementSize, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale, Op->AddrSize);
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale);
}
}
@@ -1268,9 +1247,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
const bool SupportsSVELoad = HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4) && Op->AddrSize == IR::OpSize::i64Bit;
if (SupportsSVELoad) {
if (HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4)) {
ARMEmitter::SVEModType ModType = ARMEmitter::SVEModType::MOD_NONE;
if (OffsetScale != 1) {
ModType = ARMEmitter::SVEModType::MOD_LSL;
@@ -1324,7 +1301,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
}
} else {
Emulate128BitGather(IR::OpSize::i128Bit, IR::OpSize::i32Bit, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
IR::OpSize::i64Bit, 0, 0, OffsetScale, Op->AddrSize);
IR::OpSize::i64Bit, 0, 0, OffsetScale);
}
}
@@ -1625,7 +1602,7 @@ DEF_OP(StoreMem) {
const auto MemReg = GetReg(Op->Addr);
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
switch (OpSize) {
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
@@ -1736,7 +1713,7 @@ DEF_OP(StoreMemPair) {
const auto OpSize = IROp->Size;
const auto Addr = GetReg(Op->Addr);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src1 = GetZeroableReg(Op->Value1);
const auto Src2 = GetZeroableReg(Op->Value2);
switch (OpSize) {
@@ -1763,13 +1740,13 @@ DEF_OP(StoreMemTSO) {
const auto MemReg = GetReg(Op->Addr);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
}
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
@@ -1790,7 +1767,7 @@ DEF_OP(StoreMemTSO) {
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == IR::RegClass::GPR) {
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
if (OpSize == IR::OpSize::i8Bit) {
@@ -1874,7 +1851,7 @@ DEF_OP(MemSet) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
(void)tbnz(DirectionReg, 1, &BackwardImpl);
tbnz(DirectionReg, 1, &BackwardImpl);
}
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
@@ -1922,7 +1899,7 @@ DEF_OP(MemSet) {
ARMEmitter::ForwardLabel DoneInternal {};
// Early exit if zero count.
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AgainInternal256Exit {};
@@ -1939,50 +1916,50 @@ DEF_OP(MemSet) {
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
tbnz(TMP1, 63, &AgainInternal128Exit);
// Fill VTMP2 with the set pattern
dup(SubRegSize, VTMP2.Q(), Value);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
tbnz(TMP1, 63, &AgainInternal256Exit);
(void)Bind(&AgainInternal256);
Bind(&AgainInternal256);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)tbz(TMP1, 63, &AgainInternal256);
tbz(TMP1, 63, &AgainInternal256);
(void)Bind(&AgainInternal256Exit);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
(void)Bind(&AgainInternal128);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbz(TMP1, 63, &AgainInternal128);
tbz(TMP1, 63, &AgainInternal128);
(void)Bind(&AgainInternal128Exit);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
}
}
(void)Bind(&AgainInternal);
Bind(&AgainInternal);
if (IsAtomic) {
MemStoreTSO(Value, OpSize, SizeDirection);
} else {
MemStore(Value, OpSize, SizeDirection);
}
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
(void)Bind(&DoneInternal);
Bind(&DoneInternal);
if (SizeDirection >= 0) {
switch (OpSize) {
@@ -2012,12 +1989,12 @@ DEF_OP(MemSet) {
EmitMemset(Direction);
if (Direction == 1) {
(void)b(&Done);
(void)Bind(&BackwardImpl);
b(&Done);
Bind(&BackwardImpl);
}
}
(void)Bind(&Done);
Bind(&Done);
// Destination already set to the final pointer.
}
}
@@ -2067,7 +2044,7 @@ DEF_OP(MemCpy) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
(void)tbnz(DirectionReg, 1, &BackwardImpl);
tbnz(DirectionReg, 1, &BackwardImpl);
}
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
@@ -2164,7 +2141,7 @@ DEF_OP(MemCpy) {
ARMEmitter::ForwardLabel DoneInternal {};
// Early exit if zero count.
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AbsPos {};
@@ -2174,11 +2151,11 @@ DEF_OP(MemCpy) {
ARMEmitter::BackwardLabel AgainInternal256 {};
sub(ARMEmitter::Size::i64Bit, TMP4, TMP2, TMP3);
(void)tbz(TMP4, 63, &AbsPos);
tbz(TMP4, 63, &AbsPos);
neg(ARMEmitter::Size::i64Bit, TMP4, TMP4);
(void)Bind(&AbsPos);
Bind(&AbsPos);
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
(void)tbnz(TMP4, 63, &AgainInternal);
tbnz(TMP4, 63, &AgainInternal);
if (Direction == -1) {
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
@@ -2190,30 +2167,30 @@ DEF_OP(MemCpy) {
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
tbnz(TMP1, 63, &AgainInternal128Exit);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
tbnz(TMP1, 63, &AgainInternal256Exit);
(void)Bind(&AgainInternal256);
Bind(&AgainInternal256);
MemCpy(32, 32 * Direction);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)tbz(TMP1, 63, &AgainInternal256);
tbz(TMP1, 63, &AgainInternal256);
(void)Bind(&AgainInternal256Exit);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
(void)Bind(&AgainInternal128);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbz(TMP1, 63, &AgainInternal128);
tbz(TMP1, 63, &AgainInternal128);
(void)Bind(&AgainInternal128Exit);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
@@ -2221,16 +2198,16 @@ DEF_OP(MemCpy) {
}
}
(void)Bind(&AgainInternal);
Bind(&AgainInternal);
if (IsAtomic) {
MemCpyTSO(OpSize, SizeDirection);
} else {
MemCpy(OpSize, SizeDirection);
}
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
(void)Bind(&DoneInternal);
Bind(&DoneInternal);
// Needs to use temporaries just in case of overwrite
mov(TMP1, MemRegDest.X());
@@ -2288,15 +2265,186 @@ DEF_OP(MemCpy) {
for (int32_t Direction : {1, -1}) {
EmitMemcpy(Direction);
if (Direction == 1) {
(void)b(&Done);
(void)Bind(&BackwardImpl);
b(&Done);
Bind(&BackwardImpl);
}
}
(void)Bind(&Done);
Bind(&Done);
// Destination already set to the final pointer.
}
}
DEF_OP(ParanoidLoadMemTSO) {
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
const auto OpSize = IROp->Size;
auto MemReg = GetReg(Op->Addr);
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
if (!IsInlineConstant(Op->Offset, &Offset)) {
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
}
}
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
const auto Dst = GetReg(Node);
ldapurb(Dst, MemReg, Offset);
} else {
switch (OpSize) {
case IR::OpSize::i16Bit: ldapurh(Dst, MemReg, Offset); break;
case IR::OpSize::i32Bit: ldapur(Dst.W(), MemReg, Offset); break;
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
ldaprb(Dst.W(), MemReg);
} else {
switch (OpSize) {
case IR::OpSize::i16Bit: ldaprh(Dst.W(), MemReg); break;
case IR::OpSize::i32Bit: ldapr(Dst.W(), MemReg); break;
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit: ldarb(Dst, MemReg); break;
case IR::OpSize::i16Bit: ldarh(Dst, MemReg); break;
case IR::OpSize::i32Bit: ldar(Dst.W(), MemReg); break;
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
} else {
const auto Dst = GetVReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit:
ldarb(TMP1, MemReg);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
break;
case IR::OpSize::i16Bit:
ldarh(TMP1, MemReg);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
break;
case IR::OpSize::i32Bit:
ldar(TMP1.W(), MemReg);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
break;
case IR::OpSize::i64Bit:
ldar(TMP1, MemReg);
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
break;
case IR::OpSize::i128Bit:
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
clrex();
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
break;
case IR::OpSize::i256Bit:
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
dmb(ARMEmitter::BarrierScope::ISH);
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
dmb(ARMEmitter::BarrierScope::ISH);
break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
}
DEF_OP(ParanoidStoreMemTSO) {
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
const auto OpSize = IROp->Size;
auto MemReg = GetReg(Op->Addr);
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
if (!IsInlineConstant(Op->Offset, &Offset)) {
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
}
}
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
stlurb(Src, MemReg, Offset);
} else {
switch (OpSize) {
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
case IR::OpSize::i64Bit: stlur(Src.X(), MemReg, Offset); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
case IR::OpSize::i64Bit: stlr(Src.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
} else {
const auto Src = GetVReg(Op->Value);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit:
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
stlrb(TMP1, MemReg);
break;
case IR::OpSize::i16Bit:
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
stlrh(TMP1, MemReg);
break;
case IR::OpSize::i32Bit:
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
stlr(TMP1.W(), MemReg);
break;
case IR::OpSize::i64Bit:
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
stlr(TMP1, MemReg);
break;
case IR::OpSize::i128Bit: {
// Move vector to GPRs
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
umov<ARMEmitter::SubRegSize::i64Bit>(TMP2, Src, 1);
ARMEmitter::BackwardLabel B;
Bind(&B);
// ldaxp must not have both the destination registers be the same
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
break;
}
case IR::OpSize::i256Bit: {
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
dmb(ARMEmitter::BarrierScope::ISH);
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
dmb(ARMEmitter::BarrierScope::ISH);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
}
}
DEF_OP(CacheLineClear) {
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
dmb(ARMEmitter::BarrierScope::SY);
+19 -21
View File
@@ -10,12 +10,10 @@ $end_info$
#endif
#include "Interface/Context/Context.h"
#include "Interface/Core/JIT/DebugData.h"
#include "Interface/Core/JIT/JITClass.h"
#include "FEXCore/Debug/InternalThreadState.h"
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/EnumUtils.h>
namespace FEXCore::CPU {
@@ -48,10 +46,10 @@ DEF_OP(GuestOpcode) {
DEF_OP(Fence) {
auto Op = IROp->C<IR::IROp_Fence>();
switch (Op->Fence) {
case IR::FenceType::Load: dmb(ARMEmitter::BarrierScope::LD); break;
case IR::FenceType::LoadStore: dmb(ARMEmitter::BarrierScope::SY); break;
case IR::FenceType::Store: dmb(ARMEmitter::BarrierScope::ST); break;
case IR::FenceType::Inst: isb(); break;
case IR::Fence_Load.Val: dmb(ARMEmitter::BarrierScope::LD); break;
case IR::Fence_LoadStore.Val: dmb(ARMEmitter::BarrierScope::SY); break;
case IR::Fence_Store.Val: dmb(ARMEmitter::BarrierScope::ST); break;
case IR::Fence_Inst.Val: isb(); break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
}
}
@@ -78,19 +76,19 @@ DEF_OP(Break) {
switch (Op->Reason.Signal) {
case Core::FAULT_SIGILL:
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGILL));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL));
br(TMP1);
break;
case Core::FAULT_SIGTRAP:
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGTRAP));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
break;
case Core::FAULT_SIGSEGV:
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGSEGV));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV));
br(TMP1);
break;
default:
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGTRAP));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
break;
}
@@ -108,10 +106,10 @@ DEF_OP(GetRoundingMode) {
// zero. Just swapping 01 and 10. That's a bitfield reverse. Round mode is in
// bottom two bits. After reversing as a 32-bit operation, it'll be in [31:30]
// and ripe for reinsertion back at 0.
static_assert(FEXCore::ToUnderlying(IR::RoundMode::Nearest) == 0);
static_assert(FEXCore::ToUnderlying(IR::RoundMode::NegInfinity) == 1);
static_assert(FEXCore::ToUnderlying(IR::RoundMode::PosInfinity) == 2);
static_assert(FEXCore::ToUnderlying(IR::RoundMode::TowardsZero) == 3);
static_assert(IR::ROUND_MODE_NEAREST == 0);
static_assert(IR::ROUND_MODE_NEGATIVE_INFINITY == 1);
static_assert(IR::ROUND_MODE_POSITIVE_INFINITY == 2);
static_assert(IR::ROUND_MODE_TOWARDS_ZERO == 3);
rbit(ARMEmitter::Size::i32Bit, TMP1, Dst);
bfi(ARMEmitter::Size::i64Bit, Dst, TMP1, 30, 2);
@@ -189,11 +187,11 @@ DEF_OP(Print) {
if (IsGPR(Op->Value)) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.PrintValue));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue));
} else {
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value), false);
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value), true);
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.PrintVectorValue));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
}
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
@@ -241,7 +239,7 @@ DEF_OP(ProcessorID) {
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
// Load the getcpu syscall number
#if defined(ARCHITECTURE_x86_64)
#if defined(_M_X86_64)
// Just to ensure the syscall number doesn't change if compiled for an x86_64 host.
constexpr auto GetCPUSyscallNum = 0xa8;
#else
@@ -305,20 +303,20 @@ DEF_OP(MonoBackpatcherWrite) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, TMP4);
}
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
#endif
ldr(ARMEmitter::XReg::x4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.MonoBackpatcherWrite));
ldr(ARMEmitter::XReg::x4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.MonoBackpatcherWrite));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, uint8_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
} else {
blr(ARMEmitter::Reg::r4);
}
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
strb(ARMEmitter::WReg::zr, TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
#endif
@@ -18,4 +18,18 @@ DEF_OP(RMWHandle) {
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(IROp->Args[0]));
}
DEF_OP(Swap1) {
auto Op = IROp->C<IR::IROp_Swap1>();
auto A = GetReg(Op->A), B = GetReg(Op->B);
LOGMAN_THROW_A_FMT(B == GetReg(Node), "Invariant");
mov(ARMEmitter::Size::i64Bit, TMP1, A);
mov(ARMEmitter::Size::i64Bit, A, B);
mov(ARMEmitter::Size::i64Bit, B, TMP1);
}
DEF_OP(Swap2) {
// Implemented above
}
} // namespace FEXCore::CPU
+24 -45
View File
@@ -1,100 +1,79 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::CPU {
enum class RelocationTypes : uint32_t {
enum class RelocationTypes : uint8_t {
// 8 byte literal in memory for symbol
// Aligned to struct RelocNamedSymbolLiteral
RELOC_NAMED_SYMBOL_LITERAL,
// Fixed size named thunk move
// 4 instruction constant generation
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocNamedThunkMove
RELOC_NAMED_THUNK_MOVE,
// 8 byte literal (relative to binary base address)
RELOC_GUEST_RIP_LITERAL,
// Fixed size guest RIP move
// 4 instruction constant generation
// Aligned to struct RelocGuestRIP
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocGuestRIPMove
RELOC_GUEST_RIP_MOVE,
};
struct FEX_PACKED RelocationHeader final {
// Offset to the relocated host code data
uint64_t Offset {};
struct RelocationTypeHeader final {
RelocationTypes Type;
};
struct RelocNamedSymbolLiteral final {
enum class NamedSymbol : uint32_t {
enum class NamedSymbol : uint8_t {
///< Thread specific relocations
// JIT Literal pointers
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
};
RelocationHeader Header {};
RelocationTypeHeader Header {};
NamedSymbol Symbol;
uint32_t Pad[8];
// Offset in to the code section to begin the relocation
uint64_t Offset {};
};
struct RelocNamedThunkMove final {
RelocationHeader Header {};
RelocationTypeHeader Header {};
// GPR index the constant is being moved to
uint32_t RegisterIndex;
uint8_t RegisterIndex;
// The thunk SHA256 hash
IR::SHA256Sum Symbol;
// Offset in to the code section to begin the relocation
uint64_t Offset {};
};
struct RelocGuestRIP final {
RelocationHeader Header {};
struct RelocGuestRIPMove final {
RelocationTypeHeader Header {};
// GPR index the constant is being moved to (for non-literal relocations)
// GPR index the constant is being moved to
uint8_t RegisterIndex;
char Pad[3];
// Offset in to the code section to begin the relocation
uint64_t Offset {};
// The base RIP (to be moved by the register for non-literal relocations).
// In a serialized code cache, this is relative to the binary base address.
// The unrelocated RIP that is being moved
uint64_t GuestRIP;
uint32_t pad2[6] {};
};
union Relocation {
// Clang 16 Can't default-initialize this union
static Relocation Default() {
#if __clang_major__ < 17
Relocation Ret {.Header {}};
memset(&Ret, 0, sizeof(Ret));
return Ret;
#else
return {};
#endif
}
RelocationHeader Header {};
RelocationTypeHeader Header {};
RelocNamedSymbolLiteral NamedSymbolLiteral;
// This makes our union of relocations at least 48 bytes
// It might be more efficient to not use a union
RelocNamedThunkMove NamedThunkMove;
RelocGuestRIP GuestRIP;
RelocGuestRIPMove GuestRIPMove;
};
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl&, RelocNamedSymbolLiteral::NamedSymbol);
} // namespace FEXCore::CPU
+10 -13
View File
@@ -41,7 +41,6 @@ namespace FEXCore::CPU {
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
const auto OpSize = IROp->Size; \
const auto Is256Bit = OpSize == IR::OpSize::i256Bit; \
const auto Is128Bit = OpSize == IR::OpSize::i128Bit; \
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
\
const auto Dst = GetVReg(Node); \
@@ -50,10 +49,8 @@ namespace FEXCore::CPU {
\
if (HostSupportsSVE256 && Is256Bit) { \
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
} else if (Is128Bit) { \
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
} else { \
ARMOp(Dst.D(), Vector1.D(), Vector2.D()); \
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
} \
}
@@ -747,11 +744,11 @@ DEF_OP(VFToIScalarInsert) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
switch (RoundMode) {
case IR::RoundMode::Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
case IR::RoundMode::NegInfinity: frintm(SubRegSize.Scalar, Dst, Src); break;
case IR::RoundMode::PosInfinity: frintp(SubRegSize.Scalar, Dst, Src); break;
case IR::RoundMode::TowardsZero: frintz(SubRegSize.Scalar, Dst, Src); break;
case IR::RoundMode::Host: frinti(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Negative_Infinity: frintm(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Positive_Infinity: frintp(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Towards_Zero: frintz(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Host: frinti(SubRegSize.Scalar, Dst, Src); break;
}
};
@@ -977,7 +974,7 @@ DEF_OP(LoadNamedVectorConstant) {
}
// Load the pointer.
auto GenerateMemOperand = [this](IR::OpSize OpSize, uint32_t NamedConstant, ARMEmitter::Register Base) {
const auto ConstantOffset = offsetof(FEXCore::Core::CpuStateFrame, Pointers.NamedVectorConstants[NamedConstant]);
const auto ConstantOffset = offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.NamedVectorConstants[NamedConstant]);
if (ConstantOffset <= 255 || // Unscaled 9-bit signed
((ConstantOffset & (IR::OpSizeToSize(OpSize) - 1)) == 0 &&
@@ -985,13 +982,13 @@ DEF_OP(LoadNamedVectorConstant) {
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, ConstantOffset);
}
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.NamedVectorConstantPointers[NamedConstant]));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[NamedConstant]));
return ARMEmitter::ExtendedMemOperand(TMP1, ARMEmitter::IndexType::OFFSET, 0);
};
if (OpSize == IR::OpSize::i256Bit) {
// Handle SVE 32-byte variant upfront.
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.NamedVectorConstantPointers[Op->Constant]));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[Op->Constant]));
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), TMP1, 0);
return;
}
@@ -1013,7 +1010,7 @@ DEF_OP(LoadNamedVectorIndexedConstant) {
const auto Dst = GetVReg(Node);
// Load the pointer.
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.IndexedNamedVectorConstantPointers[Op->Constant]));
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.IndexedNamedVectorConstantPointers[Op->Constant]));
switch (OpSize) {
case IR::OpSize::i8Bit: ldrb(Dst, TMP1, Op->Index); break;
+17 -24
View File
@@ -15,7 +15,7 @@ $end_info$
namespace FEXCore {
GuestToHostMap::GuestToHostMap()
: BlockLinks_mbr {"FEXMem_BlockLinks"} {
: BlockLinks_mbr {fextl::pmr::get_default_resource()} {
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
// Setup our PMR map.
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
@@ -24,7 +24,7 @@ GuestToHostMap::GuestToHostMap()
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
: ctx {CTX} {
TotalCacheSize = ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE + MAX_L1_SIZE;
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
// Block cache ends up looking like this
// PageMemoryMap[VirtualMemoryRegion >> 12]
@@ -39,10 +39,6 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
// We need one pointer per page of virtual memory
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize, false, false));
LOGMAN_THROW_A_FMT(PagePointer != -1ULL, "Failed to allocate PagePointer");
FEXCore::Allocator::VirtualName("FEXMem_Lookup", reinterpret_cast<void*>(PagePointer),
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE);
CTX->SyscallHandler->MarkOvercommitRange(PagePointer, TotalCacheSize);
// Allocate our memory backing our pages
@@ -50,21 +46,14 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
// XXX: We can drop down to 16KB if we store 4byte offsets from the code base
// We currently limit to 128MB of real memory for caching for the total cache size.
// Can end up being inefficient if we compile a small number of blocks per page
PageMemory = PagePointer + ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8;
PageMemory = PagePointer + ctx->Config.VirtualMemSize / 4096 * 8;
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
// L1 Cache
L1Pointer = PageMemory + CODE_SIZE;
FEXCore::Allocator::VirtualName("FEXMem_Lookup_L1", reinterpret_cast<void*>(L1Pointer), MAX_L1_SIZE);
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
VirtualMemSize = ctx->Config.VirtualMemSize;
if (DynamicL1Cache()) {
// Start at minimum size when dynamic.
L1PointerMask = MIN_L1_ENTRIES - 1;
} else {
// Start at maximum instead.
L1PointerMask = MAX_L1_ENTRIES - 1;
}
}
LookupCache::~LookupCache() {
@@ -75,27 +64,31 @@ LookupCache::~LookupCache() {
// These will get freed when their memory allocators are deallocated.
}
void LookupCache::ClearL2Cache(const FEXCore::LookupCacheBaseLockToken& lk) {
void LookupCache::ClearL2Cache() {
auto lk = Shared->AcquireLock();
// Clear out the page memory
// PagePointer and PageMemory are sequential with each other. Clear both at once.
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer),
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE, false);
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
AllocateOffset = 0;
}
void LookupCache::ClearThreadLocalCaches(const LookupCacheWriteLockToken&) {
void LookupCache::ClearThreadLocalCaches() {
auto lk = Shared->AcquireLock();
// Clear L1 and L2 by clearing the full cache.
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
CachedCodePages.clear();
}
void LookupCache::ClearCache(const LookupCacheWriteLockToken& lk) {
void LookupCache::ClearCache() {
auto lk = Shared->AcquireLock();
// Clear L1 and L2 by clearing the full cache.
ClearThreadLocalCaches(lk);
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
Shared->ClearCache(lk);
}
void GuestToHostMap::ClearCache(const LookupCacheWriteLockToken&) {
void GuestToHostMap::ClearCache(const LockToken&) {
// Allocate a new pointer from the BlockLinks pma again.
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
// All code is gone, clear the block list
+118 -264
View File
@@ -2,57 +2,30 @@
#pragma once
#include "Interface/Context/Context.h"
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/SHMStats.h>
#include "Utils/WritePriorityMutex.h"
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/memory_resource.h>
#include <FEXCore/fextl/robin_map.h>
#include <FEXCore/fextl/robin_set.h>
#include <FEXCore/fextl/vector.h>
#include <FEXCore/fextl/memory_resource.h>
#include <cstdint>
#include <functional>
#include <stddef.h>
#include <utility>
#include <mutex>
namespace FEXCore {
struct LookupCacheBaseLockToken {
protected:
// Protected constructor - only derived classes can construct
LookupCacheBaseLockToken() = default;
};
struct LookupCacheWriteLockToken : public LookupCacheBaseLockToken {
private:
// Only constructible by GuestToHostMap
friend struct GuestToHostMap;
LookupCacheWriteLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
: Lock {Mutex} {}
std::lock_guard<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
};
struct LookupCacheReadLockToken : public LookupCacheBaseLockToken {
private:
// Only constructible by GuestToHostMap
friend struct GuestToHostMap;
LookupCacheReadLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
: Lock {Mutex} {}
std::shared_lock<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
};
struct GuestToHostMap {
FEXCore::Utils::WritePriorityMutex::Mutex Lock {};
std::recursive_mutex WriteLock;
struct LockToken {
std::lock_guard<std::recursive_mutex> Lock;
};
[[nodiscard]]
LookupCacheWriteLockToken AcquireWriteLock() {
return LookupCacheWriteLockToken {Lock};
}
[[nodiscard]]
LookupCacheReadLockToken AcquireReadLock() {
return LookupCacheReadLockToken {Lock};
LockToken AcquireLock() {
return LockToken {std::lock_guard {WriteLock}};
}
struct BlockLinkTag {
@@ -76,72 +49,53 @@ struct GuestToHostMap {
// walking each block member and destructing objects.
//
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
fextl::pmr::named_monotonic_page_buffer_resource BlockLinks_mbr;
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
BlockLinksMapType* BlockLinks;
struct BlockEntry {
uint64_t HostCode;
fextl::vector<uint64_t> CodePages;
};
fextl::robin_map<uint64_t, BlockEntry> BlockList;
fextl::robin_map<uint64_t, uint64_t> BlockList;
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
GuestToHostMap();
// Adds to Guest -> Host code mapping
const BlockEntry& AddBlockMapping(uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
void AddBlockMapping(uint64_t Address, void* HostCode, const LockToken&) {
// This may replace an existing mapping
// NOTE: Generally no previous entry should exist, however there is one exception:
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
// may already contain the block address. Since is comparatively rare, we'll just leak
// one of the two blocks in this case.
return BlockList.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, CodePages}).first->second;
BlockList[Address] = (uintptr_t)HostCode;
}
const BlockEntry* FindBlock(uint64_t Address, const LookupCacheReadLockToken&) {
std::optional<uintptr_t> FindBlock(uint64_t Address, const LockToken&) {
auto HostCode = BlockList.find(Address);
if (HostCode == BlockList.end()) {
return nullptr;
return std::nullopt;
}
return &HostCode->second;
return HostCode->second;
}
bool Erase(uint64_t Address, const LookupCacheWriteLockToken&) {
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LockToken&) {
// Sever any links to this block
auto lower = BlockLinks->lower_bound({Address, nullptr});
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
it->second(it->first.HostLink);
it->second(Frame, it->first.HostLink);
}
// Remove from BlockList
return BlockList.erase(Address) != 0;
}
void InvalidateRange(uint64_t Start, uint64_t Length) {
auto lk = AcquireWriteLock();
auto lower = CodePages.lower_bound(Start >> 12);
auto upper = CodePages.upper_bound((Start + Length - 1) >> 12);
for (auto it = lower; it != upper; it++) {
for (const auto& Entry : it->second) {
Erase(Entry, lk);
}
}
CodePages.erase(lower, upper);
}
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken&) {
const FEXCore::Context::BlockDelinkerFunc& delinker, const LockToken&) {
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
}
bool AddBlockExecutableRange(const std::ranges::input_range auto& Addresses, uint64_t Start, uint64_t Length, const LookupCacheWriteLockToken&) {
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LockToken&) {
bool rv = false;
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
@@ -153,7 +107,7 @@ struct GuestToHostMap {
return rv;
}
void ClearCache(const LookupCacheWriteLockToken&);
void ClearCache(const LockToken&);
};
class LookupCache {
@@ -168,199 +122,122 @@ public:
// Swaps out the underlying GuestToHostMap and clears all associated caches.
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap, const LookupCacheWriteLockToken& lk) {
ClearThreadLocalCaches(lk);
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap) {
ClearThreadLocalCaches();
Shared = &NewMap;
}
uintptr_t FindBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) {
uintptr_t FindBlock(uint64_t Address) {
// Try L1, no lock needed
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
if (L1Entry.GuestCode == Address) {
return L1Entry.HostCode;
}
// L2 and L3 need to be locked
uintptr_t HostPtr {};
{
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheReadLockTime : nullptr);
auto lk = Shared->AcquireReadLock();
LockTime.reset();
auto lk = Shared->AcquireLock();
if (!DisableL2Cache()) {
// Try L2
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
const auto PageOffset = Address & (0x0FFF);
// Try L2
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
const auto PageOffset = Address & (0x0FFF);
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
auto LocalPagePointer = Pointers[PageIndex];
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
auto LocalPagePointer = Pointers[PageIndex];
// Do we a page pointer for this address?
if (LocalPagePointer) {
// Find there pointer for the address in the blocks
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
// Do we a page pointer for this address?
if (LocalPagePointer) {
// Find there pointer for the address in the blocks
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
if (BlockPointers[PageOffset].GuestCode == Address) {
L1Entry.GuestCode = Address;
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
HostPtr = L1Entry.HostCode;
}
}
}
if (!HostPtr) {
// Try L3
auto Entry = Shared->FindBlock(Address, lk);
if (Entry) {
CacheBlockMapping(Address, *Entry, false, lk);
HostPtr = Entry->HostCode;
}
if (BlockPointers[PageOffset].GuestCode == Address) {
L1Entry.GuestCode = Address;
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
return L1Entry.HostCode;
}
}
if (HostPtr && DynamicL1Cache()) {
UpdateDynamicL1Stats(Thread);
// Try L3
auto HostCode = Shared->FindBlock(Address, lk);
if (HostCode) {
CacheBlockMapping(Address, HostCode.value());
return HostCode.value();
}
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedCacheMissCount, 1);
return HostPtr;
}
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread) {
// If host pointer was found in L2 or L3, then add it to the counter.
// Keeping track not L1 misses, but specifically L2/L3 hits.
++L2L3CacheHits;
const auto CurrentTime = std::chrono::system_clock::now();
const auto Period = CurrentTime - LastPeriod;
if (Period >= SamplePeriod) {
// If larger than the sample period then check if we need to increase L1 cache size.
const double AveragePerSecond = static_cast<double>(L2L3CacheHits) /
static_cast<double>(std::chrono::duration_cast<std::chrono::milliseconds>(Period).count()) * 1000.0;
if (AveragePerSecond >= DynamicL1CacheIncreaseCountHeuristic()) {
if (CurrentL1Entries < MAX_L1_ENTRIES) {
CurrentL1Entries <<= 1;
L1PointerMask = CurrentL1Entries - 1;
// Update the thread's L1 pointer mask to increase how much cache it uses.
// Since we're in C-code, this is safe to update here.
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
}
} else if (AveragePerSecond < DynamicL1CacheDecreaseCountHeuristic()) {
if (CurrentL1Entries > MIN_L1_ENTRIES) {
CurrentL1Entries >>= 1;
L1PointerMask = CurrentL1Entries - 1;
// Madvise the entries that we are dropping. Gives the memory back to the OS.
LookupCacheEntry* FirstZeroL1Entry = &reinterpret_cast<LookupCacheEntry*>(L1Pointer)[CurrentL1Entries];
size_t ZeroMemorySize = (MAX_L1_ENTRIES - CurrentL1Entries) * sizeof(LookupCacheEntry);
FEXCore::Allocator::VirtualDontNeed(FirstZeroL1Entry, ZeroMemorySize, false);
// Update the thread's L1 pointer mask to increase how much cache it uses.
// Since we're in C-code, this is safe to update here.
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
}
}
// Update Last period to start again.
LastPeriod = CurrentTime;
L2L3CacheHits = 0;
}
// Failed to find
return 0;
}
GuestToHostMap* Shared = nullptr;
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
// Returns true if new pages are marked as containing code
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
auto lk = Shared->AcquireWriteLock();
LockTime.reset();
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
auto lk = Shared->AcquireLock();
return Shared->AddBlockExecutableRange(Addresses, Start, Length, lk);
}
// Adds to Guest -> Host code mapping
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode) {
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
auto lk = Shared->AcquireWriteLock();
LockTime.reset();
void AddBlockMapping(uint64_t Address, void* HostCode) {
auto lk = Shared->AcquireLock();
const auto& Entry = Shared->AddBlockMapping(Address, CodePages, HostCode, lk);
Shared->AddBlockMapping(Address, HostCode, lk);
// There is no need to update L1 or L2, they will get updated on first lookup
// However, adding to L1 here increases performance
CacheBlockMapping(Address, Entry, true, lk);
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
L1Entry.GuestCode = Address;
L1Entry.HostCode = (uintptr_t)HostCode;
}
// Invalidates L1/L2 for a given guest block
void InvalidateCache(uint64_t Address, const LookupCacheWriteLockToken& lk) {
// NOTE: It's the caller's responsibility to call Erase() for all other
// GuestToHostMaps that share the same LookupCache. Otherwise, the
// L1/L2 caches will contain stale references to deallocated memory.
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
auto lk = Shared->AcquireLock();
bool ErasedAny = Shared->Erase(Frame, Address, lk);
// Do L1
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
if (L1Entry.GuestCode == Address) {
L1Entry.GuestCode = 0;
ErasedAny = true;
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
// This is a soft guarantee for cross thread invalidation, as atomics are not used
// and it hasn't been thoroughly tested
}
if (!DisableL2Cache()) {
// Do full map
Address = Address & (VirtualMemSize - 1);
uint64_t PageOffset = Address & (0x0FFF);
Address >>= 12;
// Do full map
Address = Address & (VirtualMemSize - 1);
uint64_t PageOffset = Address & (0x0FFF);
Address >>= 12;
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uint64_t LocalPagePointer = Pointers[Address];
if (!LocalPagePointer) {
// Page for this code didn't even exist, nothing to do
return;
}
// Page exists, just set the offset to zero
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
BlockPointers[PageOffset].GuestCode = 0;
BlockPointers[PageOffset].HostCode = 0;
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uint64_t LocalPagePointer = Pointers[Address];
if (!LocalPagePointer) {
// Page for this code didn't even exist, nothing to do
return ErasedAny;
}
// Page exists, just set the offset to zero
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
BlockPointers[PageOffset].GuestCode = 0;
BlockPointers[PageOffset].HostCode = 0;
return true;
}
// Invalidates all L1/L2 entries for all guest block that intersect the given range
bool InvalidateCacheRange(uint64_t Start, uint64_t Length) {
auto lk = Shared->AcquireWriteLock();
auto lower = CachedCodePages.lower_bound(Start >> 12);
auto upper = CachedCodePages.upper_bound((Start + Length - 1) >> 12);
for (auto it = lower; it != upper; it++) {
for (const auto& Entry : it->second) {
InvalidateCache(Entry, lk);
}
}
bool ret = upper != lower;
CachedCodePages.erase(lower, upper);
return ret;
}
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken& lk) {
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
auto lk = Shared->AcquireLock();
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
}
void ClearCache(const LookupCacheWriteLockToken&);
void ClearL2Cache(const LookupCacheBaseLockToken&);
void ClearThreadLocalCaches(const LookupCacheWriteLockToken&);
void ClearCache();
void ClearL2Cache();
void ClearThreadLocalCaches();
uintptr_t GetL1Pointer() const {
return L1Pointer;
}
uintptr_t GetScaledL1PointerMask() const {
return L1PointerMask << FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry));
}
uintptr_t GetPagePointer() const {
return PagePointer;
}
@@ -368,6 +245,9 @@ public:
return VirtualMemSize;
}
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
// This needs to be taken before reads or writes to L2, L3, CodePages,
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
// may only happen during cross thread invalidation (::Erase).
@@ -375,52 +255,45 @@ public:
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
// This approach has not been fully vetted yet.
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
auto AcquireWriteLock() {
return Shared->AcquireWriteLock();
auto AcquireLock() {
return Shared->AcquireLock();
}
private:
void CacheBlockMapping(uint64_t Address, const GuestToHostMap::BlockEntry& Entry, bool L1Only, const LookupCacheBaseLockToken& lk) {
for (const auto& CodePage : Entry.CodePages) {
CachedCodePages[CodePage >> 12].insert(Address);
}
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
// Do L1
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
L1Entry.GuestCode = Address;
L1Entry.HostCode = Entry.HostCode;
L1Entry.HostCode = HostCode;
if (!DisableL2Cache() && !L1Only) {
// Do ful map
auto FullAddress = Address;
Address = Address & (VirtualMemSize - 1);
// Do ful map
auto FullAddress = Address;
Address = Address & (VirtualMemSize - 1);
uint64_t PageOffset = Address & (0x0FFF);
Address >>= 12;
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uint64_t LocalPagePointer = Pointers[Address];
if (!LocalPagePointer) {
// We don't have a page pointer for this address
// Allocate one now if we can
uintptr_t NewPageBacking = AllocateBackingForPage();
if (!NewPageBacking) {
// Couldn't allocate, clear L2 and retry
ClearL2Cache(lk);
CacheBlockMapping(FullAddress, Entry, false, lk);
return;
}
Pointers[Address] = NewPageBacking;
LocalPagePointer = NewPageBacking;
uint64_t PageOffset = Address & (0x0FFF);
Address >>= 12;
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uint64_t LocalPagePointer = Pointers[Address];
if (!LocalPagePointer) {
// We don't have a page pointer for this address
// Allocate one now if we can
uintptr_t NewPageBacking = AllocateBackingForPage();
if (!NewPageBacking) {
// Couldn't allocate, clear L2 and retry
ClearL2Cache();
CacheBlockMapping(Address, HostCode);
return;
}
// Add the new pointer to the page block
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
// This silently replaces existing mappings
BlockPointers[PageOffset].GuestCode = FullAddress;
BlockPointers[PageOffset].HostCode = Entry.HostCode;
Pointers[Address] = NewPageBacking;
LocalPagePointer = NewPageBacking;
}
// Add the new pointer to the page block
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
// This silently replaces existing mappings
BlockPointers[PageOffset].GuestCode = FullAddress;
BlockPointers[PageOffset].HostCode = HostCode;
}
uintptr_t AllocateBackingForPage() {
@@ -437,38 +310,19 @@ private:
return PageMemory + NewBase;
}
// Maps from a page index to all blocks in the page that have at some point been fetched into L1/L2
fextl::map<uint64_t, fextl::robin_set<uint64_t>> CachedCodePages;
uintptr_t PagePointer;
uintptr_t PageMemory;
uintptr_t L1Pointer;
uintptr_t L1PointerMask;
size_t TotalCacheSize;
// Start with 8k entries in L1 to give 128KB of L1 cache to each thread.
// Max out at 1 million entries to give each thread 16MB of L1 cache maximum.
constexpr static size_t MIN_L1_ENTRIES = 8 * 1024; // Must be a power of 2
constexpr static size_t MAX_L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
constexpr static size_t SIZE_PER_PAGE = FEXCore::Utils::FEX_PAGE_SIZE * sizeof(LookupCacheEntry);
constexpr static size_t MAX_L1_SIZE = MAX_L1_ENTRIES * sizeof(LookupCacheEntry);
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
constexpr static size_t L1_SIZE = L1_ENTRIES * sizeof(LookupCacheEntry);
size_t AllocateOffset {};
FEXCore::Context::ContextImpl* ctx;
uint64_t VirtualMemSize {};
size_t CurrentL1Entries = MIN_L1_ENTRIES;
uint64_t L2L3CacheHits {};
std::chrono::time_point<std::chrono::system_clock> LastPeriod {};
constexpr static std::chrono::seconds SamplePeriod {1};
FEX_CONFIG_OPT(DynamicL1CacheIncreaseCountHeuristic, DYNAMICL1CACHEINCREASECOUNTHEURISTIC);
FEX_CONFIG_OPT(DynamicL1CacheDecreaseCountHeuristic, DYNAMICL1CACHEDECREASECOUNTHEURISTIC);
FEX_CONFIG_OPT(DynamicL1Cache, DYNAMICL1CACHE);
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
};
} // namespace FEXCore
File diff suppressed because it is too large. Load diff
+125 -208
View File
@@ -139,28 +139,27 @@ public:
FlushRegisterCache();
return _Jump(_TargetBlock);
}
IRPair<IROp_CondJump> CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClass _Cond = CondClass::NEQ,
IRPair<IROp_CondJump> CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClassType _Cond = {COND_NEQ},
IR::OpSize _CompareSize = OpSize::iInvalid) {
FlushRegisterCache();
return _CondJump(_Cmp1, _Cmp2, _TrueBlock, _FalseBlock, _Cond, _CompareSize);
}
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClass cond = CondClass::NEQ) {
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
FlushRegisterCache();
return _CondJump(ssa0, cond);
}
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClass cond = CondClass::NEQ) {
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
FlushRegisterCache();
return _CondJump(ssa0, ssa1, ssa2, cond);
}
IRPair<IROp_CondJump> CondJumpNZCV(CondClass Cond) {
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
FlushRegisterCache();
return _CondJump(InvalidNode, InvalidNode, InvalidNode, InvalidNode, Cond, OpSize::iInvalid, true);
}
IRPair<IROp_CondJump> CondJumpBit(Ref Src, unsigned Bit, bool Set) {
FlushRegisterCache();
auto InlineConst = _InlineConstant(Bit);
auto Cond = Set ? CondClass::TSTNZ : CondClass::TSTZ;
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, Cond, OpSize::iInvalid, false);
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, {Set ? COND_TSTNZ : COND_TSTZ}, OpSize::iInvalid, false);
}
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint = BranchHint::None) {
FlushRegisterCache();
@@ -210,17 +209,10 @@ public:
}
static bool CanHaveSideEffects(const FEXCore::X86Tables::X86InstInfo* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
if (TableInfo) {
if (TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
// If it is marked as having memory access then always say it has a side-effect.
// Not always true but better to be safe.
return true;
}
if (TableInfo->Flags & (X86Tables::InstFlags::FLAGS_SETS_RIP | X86Tables::InstFlags::FLAGS_BLOCK_END)) {
// Cooperative suspend interrupts can be triggered at any back-edge, the RIP must be reconstructed correctly in such cases
return true;
}
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
// If it is marked as having memory access then always say it has a side-effect.
// Not always true but better to be safe.
return true;
}
auto CanHaveSideEffects = false;
@@ -252,7 +244,7 @@ public:
auto ExitBlock = CreateNewCodeBlockAfter(BackwardBlock);
auto DF = GetRFLAG(X86State::RFLAG_DF_RAW_LOC);
CondJump(DF, Zero, ForwardBlock, BackwardBlock, CondClass::EQ);
CondJump(DF, Zero, ForwardBlock, BackwardBlock, {COND_EQ});
for (auto D = 0; D < 2; ++D) {
SetCurrentCodeBlock(D ? BackwardBlock : ForwardBlock);
@@ -301,8 +293,7 @@ public:
return ShouldDump;
}
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions,
bool Is64BitMode, bool MonoBackpatcherBlock);
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions, bool Is64BitMode, bool MonoBackpatcherBlock);
void Finalize();
// Dispatch builder functions
@@ -372,7 +363,7 @@ public:
void CMOVOp(OpcodeArgs);
void CPUIDOp(OpcodeArgs);
void XGetBVOp(OpcodeArgs);
uint32_t GetConstantShift(X86Tables::DecodedOp Op, bool Is1Bit);
uint32_t LoadConstantShift(X86Tables::DecodedOp Op, bool Is1Bit);
void SHLOp(OpcodeArgs);
void SHLImmediateOp(OpcodeArgs, bool SHL1Bit);
void SHROp(OpcodeArgs);
@@ -564,7 +555,7 @@ public:
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
RoundMode TranslateRoundType(uint8_t Mode);
RoundType TranslateRoundType(uint8_t Mode);
template<IR::OpSize ElementSize>
void InsertScalarRound(OpcodeArgs);
@@ -758,6 +749,7 @@ public:
void X87FXTRACT(OpcodeArgs);
void X87FYL2X(OpcodeArgs, bool IsFYL2XP1);
void X87LDENV(OpcodeArgs);
void X87LDSW(OpcodeArgs);
void X87ModifySTP(OpcodeArgs, bool Inc);
void X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2);
@@ -906,10 +898,6 @@ public:
void VPCLMULQDQOp(OpcodeArgs);
void CRC32(OpcodeArgs);
void Extrq_imm(OpcodeArgs);
void Insertq_imm(OpcodeArgs);
void Extrq(OpcodeArgs);
void Insertq(OpcodeArgs);
void BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDefinition);
void UnimplementedOp(OpcodeArgs);
@@ -988,6 +976,7 @@ public:
void AVX128_VPSIGN(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IROps IROp, IR::OpSize ElementSize);
Ref AVX128_VFCMPImpl(IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
void AVX128_VFCMP(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_MOVBetweenGPR_FPR(OpcodeArgs);
@@ -1006,7 +995,9 @@ public:
void AVX128_VINSERT(OpcodeArgs);
void AVX128_VINSERTPS(OpcodeArgs);
Ref AVX128_PHSUBImpl(Ref Src1, Ref Src2, size_t ElementSize);
void AVX128_VPHSUB(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_VPHSUBSW(OpcodeArgs);
void AVX128_VADDSUBP(OpcodeArgs, IR::OpSize ElementSize);
@@ -1098,8 +1089,8 @@ public:
void AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
void AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
RefPair AVX128_VPGatherQPSImpl(OpcodeArgs, Ref Dest, Ref Mask, RefVSIB VSIB);
RefPair AVX128_VPGatherImpl(OpcodeArgs, OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
RefPair AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB);
RefPair AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
void AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize);
@@ -1109,8 +1100,8 @@ public:
// End of AVX 128-bit implementation
// AVX 256-bit operations
void StoreResult_WithAVXInsert(VectorOpType Type, RegClass Class, FEXCore::X86Tables::DecodedOp Op, Ref Value,
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
void StoreResult_WithAVXInsert(VectorOpType Type, FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, Ref Value,
IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
if (Op->Dest.IsGPR() && Op->Dest.Data.GPR.GPR >= X86State::REG_XMM_0 && Op->Dest.Data.GPR.GPR <= X86State::REG_XMM_15 &&
GetGuestVectorLength() == OpSize::i256Bit && Type == VectorOpType::SSE) {
const auto gpr = Op->Dest.Data.GPR.GPR;
@@ -1161,7 +1152,7 @@ public:
}
}
void StoreContextHelper(IR::OpSize Size, RegClass Class, Ref Value, uint32_t Offset) {
void StoreContextHelper(IR::OpSize Size, RegisterClassType Class, Ref Value, uint32_t Offset) {
// For i128Bit, we won't see a normal Constant to inline, but as a special
// case we can replace with a 2x64-bit store which can use inline zeroes.
if (Size == OpSize::i128Bit) {
@@ -1173,7 +1164,7 @@ public:
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
Ref Zero = _Constant(0);
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, RegClass::GPR, Zero, Zero, Offset);
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Offset);
// XXX: This works around InlineConstant not having an associated
// register class, else we'd just do InlineConstant above.
@@ -1229,16 +1220,16 @@ public:
if (Index >= GPR0Index && Index <= GPR15Index) {
Ref R = _StoreRegister(Value, GPRSize);
R->Reg = PhysicalRegister(RegClass::GPRFixed, Index - GPR0Index).Raw;
R->Reg = PhysicalRegister(GPRFixedClass, Index - GPR0Index).Raw;
} else if (Index == PFIndex) {
_StorePF(Value, GPRSize);
} else if (Index == AFIndex) {
_StoreAF(Value, GPRSize);
} else if (Index >= FPR0Index && Index <= FPR15Index) {
Ref R = _StoreRegister(Value, VectorSize);
R->Reg = PhysicalRegister(RegClass::FPRFixed, Index - FPR0Index).Raw;
R->Reg = PhysicalRegister(FPRFixedClass, Index - FPR0Index).Raw;
} else if (Index == DFIndex) {
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(Core::CPUState, flags[X86State::RFLAG_DF_RAW_LOC]));
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(Core::CPUState, flags[X86State::RFLAG_DF_RAW_LOC]));
} else {
bool Partial = RegCache.Partial & (1ull << Index);
auto Size = Partial ? OpSize::i64Bit : CacheIndexToOpSize(Index);
@@ -1263,7 +1254,7 @@ public:
StoreContextHelper(Size, Class, Value, Offset);
// If Partial and MMX register, then we need to store all 1s in bits 64-80
if (Partial && Index >= MM0Index && Index <= MM7Index) {
_StoreContextGPR(OpSize::i16Bit, Constant(0xFFFF), Offset + 8);
_StoreContext(OpSize::i16Bit, IR::GPRClass, Constant(0xFFFF), Offset + 8);
}
}
}
@@ -1543,7 +1534,7 @@ private:
[[nodiscard]]
static bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
// Literals are immediates as sources but memory addresses as destinations.
return !(Load && (Operand.IsLiteral() || Operand.IsLiteralRelocation())) && !Operand.IsGPR();
return !(Load && Operand.IsLiteral()) && !Operand.IsGPR();
}
[[nodiscard]]
@@ -1553,63 +1544,23 @@ private:
AddressMode DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, MemoryAccessType AccessType, bool IsLoad);
Ref LoadSource(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
Ref LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
const LoadSourceOptions& Options = {});
Ref LoadSourceGPR(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
const LoadSourceOptions& Options = {}) {
return LoadSource(RegClass::GPR, Op, Operand, Flags, Options);
}
Ref LoadSourceFPR(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
const LoadSourceOptions& Options = {}) {
return LoadSource(RegClass::FPR, Op, Operand, Flags, Options);
}
Ref LoadSource_WithOpSize(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize,
uint32_t Flags, const LoadSourceOptions& Options = {});
Ref LoadSourceGPR_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize, uint32_t Flags,
const LoadSourceOptions& Options = {}) {
return LoadSource_WithOpSize(RegClass::GPR, Op, Operand, OpSize, Flags, Options);
}
Ref LoadSourceFPR_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize, uint32_t Flags,
const LoadSourceOptions& Options = {}) {
return LoadSource_WithOpSize(RegClass::FPR, Op, Operand, OpSize, Flags, Options);
}
void StoreResult_WithOpSize(RegClass Class, X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResultGPR_WithOpSize(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult_WithOpSize(RegClass::GPR, Op, Operand, Src, OpSize, Align, AccessType);
}
void StoreResultFPR_WithOpSize(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult_WithOpSize(RegClass::FPR, Op, Operand, Src, OpSize, Align, AccessType);
}
void StoreResult(RegClass Class, X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align,
Ref LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
IR::OpSize OpSize, uint32_t Flags, const LoadSourceOptions& Options = {});
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op,
const FEXCore::X86Tables::DecodedOperand& Operand, const Ref Src, IR::OpSize OpSize, IR::OpSize Align,
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand,
const Ref Src, IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const Ref Src, IR::OpSize Align,
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResultGPR(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align = OpSize::iInvalid,
MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult(RegClass::GPR, Op, Operand, Src, Align, AccessType);
}
void StoreResultFPR(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align = OpSize::iInvalid,
MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult(RegClass::FPR, Op, Operand, Src, Align, AccessType);
}
void StoreResult(RegClass Class, X86Tables::DecodedOp Op, Ref Src, OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResultGPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align = OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult(RegClass::GPR, Op, Src, Align, AccessType);
}
void StoreResultFPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align = OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult(RegClass::FPR, Op, Src, Align, AccessType);
}
// In several instances, it's desirable to get a base address with the segment offset
// applied to it. This pulls all the common-case appending into a single set of functions.
[[nodiscard]]
Ref MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize) {
Ref Mem = LoadSourceGPR_WithOpSize(Op, Operand, OpSize, Op->Flags, {.LoadData = false});
Ref Mem = LoadSource_WithOpSize(GPRClass, Op, Operand, OpSize, Op->Flags, {.LoadData = false});
return AppendSegmentOffset(Mem, Op->Flags);
}
[[nodiscard]]
@@ -1654,9 +1605,6 @@ private:
return IR::SizeToOpSize(GetSrcSize(Op));
}
[[nodiscard]]
IR::OpSize GetStringOpSize(X86Tables::DecodedOp Op) const;
// Set flag tracking to prepare for an operation that directly writes NZCV.
void HandleNZCVWrite() {
CachedNZCV = nullptr;
@@ -1848,15 +1796,14 @@ private:
// For DF, we need to transform 0/1 into 1/-1
StoreDF(_SubShift(OpSize::i64Bit, Constant(1), Value, ShiftType::LSL, 1));
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
auto PackedTF = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
auto PackedTF = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
// unblocked bit before raising the exception.
auto NewPackedTF =
_Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
_StoreContextGPR(OpSize::i8Bit, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
auto NewPackedTF = _Select(FEXCore::IR::COND_EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
_StoreContext(OpSize::i8Bit, GPRClass, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
} else {
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
}
}
@@ -1882,12 +1829,12 @@ private:
}
[[nodiscard]]
static CondClass CondForNZCVBit(unsigned BitOffset, bool Invert) {
static CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
switch (BitOffset) {
case X86State::RFLAG_SF_RAW_LOC: return Invert ? CondClass::PL : CondClass::MI;
case X86State::RFLAG_ZF_RAW_LOC: return Invert ? CondClass::NEQ : CondClass::EQ;
case X86State::RFLAG_CF_RAW_LOC: return Invert ? CondClass::ULT : CondClass::UGE;
case X86State::RFLAG_OF_RAW_LOC: return Invert ? CondClass::FNU : CondClass::FU;
case X86State::RFLAG_SF_RAW_LOC: return {Invert ? COND_PL : COND_MI};
case X86State::RFLAG_ZF_RAW_LOC: return {Invert ? COND_NEQ : COND_EQ};
case X86State::RFLAG_CF_RAW_LOC: return {Invert ? COND_ULT : COND_UGE};
case X86State::RFLAG_OF_RAW_LOC: return {Invert ? COND_FNU : COND_FU};
default: FEX_UNREACHABLE;
}
}
@@ -1898,10 +1845,10 @@ private:
static const int PFIndex = 16;
static const int AFIndex = 17;
/* Gap 18..19 */
/* Note this range is only valid if MMXState = MMXState_MMX */
static const int MM0Index = 20;
static const int MM7Index = 27;
/* Gap 28..30 */
static const int AbridgedFTWIndex = 28;
/* Gap 29..30 */
static const int DFIndex = 31;
static const int FPR0Index = 32;
static const int FPR15Index = 47;
@@ -1913,16 +1860,17 @@ private:
switch (Index) {
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
default: return ~0U;
}
}
[[nodiscard]]
static RegClass CacheIndexClass(int Index) {
static RegisterClassType CacheIndexClass(int Index) {
if ((Index >= MM0Index && Index <= MM7Index) || Index >= FPR0Index) {
return RegClass::FPR;
return FPRClass;
} else {
return RegClass::GPR;
return GPRClass;
}
}
@@ -1954,14 +1902,14 @@ private:
RegCache.Written &= ~Bit;
}
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegClass Class, IR::OpSize Size) {
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
LOGMAN_THROW_A_FMT(Index < 64, "valid index");
uint64_t Bit = (1ull << (uint64_t)Index);
if (Size == OpSize::i128Bit && (RegCache.Partial & Bit)) {
// We need to load the full register extend if we previously did a partial access.
Ref Value = RegCache.Value[Index];
Ref Full = _LoadContext(Size, Class, Offset);
Ref Full = _LoadContext(Size, RegClass, Offset);
// If we did a partial store, we're inserting into the full register
if (RegCache.Written & Bit) {
@@ -1974,8 +1922,8 @@ private:
if (!(RegCache.Cached & Bit)) {
if (Index == DFIndex) {
RegCache.Value[Index] = _LoadDF();
} else if ((Index >= MM0Index && Index <= MM7Index) || Index >= AVXHigh0Index) {
RegCache.Value[Index] = _LoadContext(Size, Class, Offset);
} else if ((Index >= MM0Index && Index <= AbridgedFTWIndex) || Index >= AVXHigh0Index) {
RegCache.Value[Index] = _LoadContext(Size, RegClass, Offset);
// We may have done a partial load, this requires special handling.
if (Size == OpSize::i64Bit) {
@@ -1986,7 +1934,7 @@ private:
} else if (Index == AFIndex) {
RegCache.Value[Index] = _LoadAF(Size);
} else {
RegCache.Value[Index] = _LoadRegister(Offset, Class, Size);
RegCache.Value[Index] = _LoadRegister(Offset, RegClass, Size);
}
RegCache.Cached |= Bit;
@@ -1995,21 +1943,21 @@ private:
return RegCache.Value[Index];
}
RefPair AllocatePair(RegClass Class, IR::OpSize Size) {
if (Class == RegClass::FPR) {
RefPair AllocatePair(FEXCore::IR::RegisterClassType Class, IR::OpSize Size) {
if (Class == FPRClass) {
return {_AllocateFPR(Size, Size), _AllocateFPR(Size, Size)};
} else {
return {_AllocateGPR(false), _AllocateGPR(false)};
}
}
RefPair LoadContextPair_Uncached(RegClass Class, IR::OpSize Size, unsigned Offset) {
RefPair LoadContextPair_Uncached(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, unsigned Offset) {
RefPair Values = AllocatePair(Class, Size);
_LoadContextPair(Size, Class, Offset, Values.Low, Values.High);
return Values;
}
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegClass Class, IR::OpSize Size) {
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
LOGMAN_THROW_A_FMT(Index != DFIndex, "must be pairable");
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
@@ -2017,7 +1965,7 @@ private:
uint64_t Bits = (3ull << (uint64_t)Index);
const auto SizeInt = IR::OpSizeToSize(Size);
if (((RegCache.Partial | RegCache.Cached) & Bits) == 0 && ((Offset / SizeInt) < 64)) {
auto Values = LoadContextPair_Uncached(Class, Size, Offset);
auto Values = LoadContextPair_Uncached(RegClass, Size, Offset);
RegCache.Value[Index] = Values.Low;
RegCache.Value[Index + 1] = Values.High;
RegCache.Cached |= Bits;
@@ -2029,13 +1977,13 @@ private:
// Fallback on a pair of loads
return {
.Low = LoadRegCache(Offset, Index, Class, Size),
.High = LoadRegCache(Offset + SizeInt, Index + 1, Class, Size),
.Low = LoadRegCache(Offset, Index, RegClass, Size),
.High = LoadRegCache(Offset + SizeInt, Index + 1, RegClass, Size),
};
}
Ref LoadGPR(uint8_t Reg) {
return LoadRegCache(Reg, GPR0Index + Reg, RegClass::GPR, GetGPROpSize());
return LoadRegCache(Reg, GPR0Index + Reg, GPRClass, GetGPROpSize());
}
Ref LoadContext(IR::OpSize Size, uint8_t Index) {
@@ -2051,7 +1999,7 @@ private:
}
Ref LoadXMMRegister(uint8_t Reg) {
return LoadRegCache(Reg, FPR0Index + Reg, RegClass::FPR, GetGuestVectorLength());
return LoadRegCache(Reg, FPR0Index + Reg, FPRClass, GetGuestVectorLength());
}
Ref LoadDF() {
@@ -2106,7 +2054,7 @@ private:
// Recover the sign bit, it is the logical DF value
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63));
} else {
return _LoadContextGPR(OpSize::i8Bit, offsetof(Core::CPUState, flags[BitOffset]));
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(Core::CPUState, flags[BitOffset]));
}
}
@@ -2123,18 +2071,18 @@ private:
}
// Safe version of NZCVSelect that handles inverted carries automatically.
Ref NZCVSelect(OpSize OpSize, CondClass Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
Ref NZCVSelect(OpSize OpSize, CondClassType Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
switch (Cond) {
case CondClass::UGE: /* cs */
case CondClass::ULT: /* cc */
case IR::COND_UGE: /* cs */
case IR::COND_ULT: /* cc */
// Invert the condition to match our expectations.
if (CarryIsInverted != CFInverted) {
Cond = (Cond == CondClass::UGE) ? CondClass::ULT : CondClass::UGE;
Cond = {Cond == COND_UGE ? COND_ULT : COND_UGE};
}
break;
case CondClass::UGT: /* hi */
case CondClass::ULE: /* ls */
case IR::COND_UGT: /* hi */
case IR::COND_ULE: /* ls */
// No clever optimization we can do here, rectify carry itself.
RectifyCarryInvert(CarryIsInverted);
break;
@@ -2223,7 +2171,7 @@ private:
HandleNZCV_RMW();
CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF, CFInverted));
StoreResultGPR(Op, Result);
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
}
// Helper to derive Dest by a given builder-using Expression with the opcode
@@ -2296,7 +2244,8 @@ private:
CachedIndexedNamedVectorConstants.clear();
}
std::optional<CondClass> DecodeNZCVCondition(uint8_t OP);
std::optional<CondClassType> DecodeNZCVCondition(uint8_t OP);
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
Ref SelectCC0All1(uint8_t OP);
/**
@@ -2310,8 +2259,8 @@ private:
if (Size != OpSize::i32Bit) {
return;
}
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
StoreResultGPR(Op, Dest);
auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
}
using ZeroShiftFunctionPtr = void (OpDispatchBuilder::*)(FEXCore::X86Tables::DecodedOp Op);
@@ -2346,7 +2295,7 @@ private:
///< Jump to zeroshift block or end block depending on if it was provided.
IRPair<IROp_CodeBlock> TailHandling = ZeroShiftResult ? ZeroShiftBlock : EndBlock;
CondJump(Shift, Zero, TailHandling, SetBlock, CondClass::EQ);
CondJump(Shift, Zero, TailHandling, SetBlock, {COND_EQ});
SetCurrentCodeBlock(SetBlock);
StartNewBlock();
@@ -2389,7 +2338,9 @@ private:
void CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High);
void CalculateFlags_UMUL(Ref High);
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res);
void CalculateFlags_ShiftLeft(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
void CalculateFlags_ShiftRight(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
void CalculateFlags_ShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
void CalculateFlags_ShiftRightDoubleImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
void CalculateFlags_ShiftRightImmediateCommon(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
@@ -2405,8 +2356,8 @@ private:
void ChgStateX87_MMX() override {
LOGMAN_THROW_A_FMT(MMXState == MMXState_X87, "Expected state to be x87");
_StackForceSlow();
SetX87Top(Constant(0)); // top reset to zero
_StoreContextGPR(OpSize::i8Bit, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
SetX87Top(Constant(0)); // top reset to zero
StoreContext(AbridgedFTWIndex, Constant(0xFFFFUL)); // all valid
MMXState = MMXState_MMX;
}
@@ -2443,62 +2394,44 @@ private:
IROp_IRHeader* CurrentHeader {};
[[nodiscard]]
bool IsTSOEnabled(RegClass Class) const {
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) const {
if (ForceTSO == ForceTSOMode::ForceEnabled) {
return true;
} else if (ForceTSO == ForceTSOMode::ForceDisabled) {
return false;
} else if (Class == RegClass::FPR) {
} else if (Class == FPRClass) {
return CTX->IsVectorAtomicTSOEnabled();
} else {
return CTX->IsAtomicTSOEnabled();
}
}
Ref _StoreMemAutoTSO(RegClass Class, OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
if (IsTSOEnabled(Class)) {
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
} else {
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
}
Ref _StoreMemGPRAutoTSO(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMemAutoTSO(RegClass::GPR, Size, Addr, Value, Align);
}
Ref _StoreMemFPRAutoTSO(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMemAutoTSO(RegClass::FPR, Size, Addr, Value, Align);
}
Ref _LoadMemAutoTSO(RegClass Class, OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = IR::OpSize::i8Bit) {
if (IsTSOEnabled(Class)) {
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
} else {
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
}
Ref _LoadMemGPRAutoTSO(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
return _LoadMemAutoTSO(RegClass::GPR, Size, ssa0, Align);
}
Ref _LoadMemFPRAutoTSO(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
return _LoadMemAutoTSO(RegClass::FPR, Size, ssa0, Align);
}
Ref _LoadMemAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
const auto B = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != RegClass::GPR, Size);
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
if (AtomicTSO) {
return _LoadMemTSO(Class, Size, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
} else {
return _LoadMem(Class, Size, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
return _LoadMem(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
}
}
Ref _LoadMemGPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
return _LoadMemAutoTSO(RegClass::GPR, Size, A, Align);
}
Ref _LoadMemFPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
return _LoadMemAutoTSO(RegClass::FPR, Size, A, Align);
}
AddressMode SelectPairAddressMode(AddressMode A, IR::OpSize Size) {
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
@@ -2516,72 +2449,56 @@ private:
}
RefPair LoadMemPair(RegClass Class, OpSize Size, Ref Base, uint32_t Offset) {
RefPair LoadMemPair(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Base, unsigned Offset) {
RefPair Values = AllocatePair(Class, Size);
_LoadMemPair(Class, Size, Base, Offset, Values.Low, Values.High);
return Values;
}
RefPair LoadMemPairFPR(OpSize Size, Ref Base, uint32_t Offset) {
return LoadMemPair(RegClass::FPR, Size, Base, Offset);
}
RefPair _LoadMemPairAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
RefPair _LoadMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
// Use ldp if possible, otherwise fallback on two loads.
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit && Size <= OpSize::i128Bit) {
const auto B = SelectPairAddressMode(A, Size);
return LoadMemPair(Class, Size, B.Base, B.Offset);
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
A = SelectPairAddressMode(A, Size);
return LoadMemPair(Class, Size, A.Base, A.Offset);
} else {
AddressMode HighA = A;
HighA.Offset += 16;
return {
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
};
}
AddressMode HighA = A;
HighA.Offset += 16;
return {
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
};
}
RefPair _LoadMemPairFPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
return _LoadMemPairAutoTSO(RegClass::FPR, Size, A, Align);
}
Ref _StoreMemAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
const auto B = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != RegClass::GPR, Size);
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
if (AtomicTSO) {
return _StoreMemTSO(Class, Size, Value, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
} else {
return _StoreMem(Class, Size, Value, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
return _StoreMem(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
}
}
Ref _StoreMemGPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMemAutoTSO(RegClass::GPR, Size, A, Value, Align);
}
Ref _StoreMemFPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMemAutoTSO(RegClass::FPR, Size, A, Value, Align);
}
void _StoreMemPairAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, Ref Value1, Ref Value2, OpSize Align = OpSize::i8Bit) {
void _StoreMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value1, Ref Value2,
IR::OpSize Align = IR::OpSize::i8Bit) {
const auto SizeInt = IR::OpSizeToSize(Size);
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
// Use stp if possible, otherwise fallback on two stores.
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit && Size <= OpSize::i128Bit) {
const auto B = SelectPairAddressMode(A, Size);
_StoreMemPair(Class, Size, Value1, Value2, B.Base, B.Offset);
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
A = SelectPairAddressMode(A, Size);
_StoreMemPair(Class, Size, Value1, Value2, A.Base, A.Offset);
} else {
auto B = A;
_StoreMemAutoTSO(Class, Size, B, Value1, OpSize::i8Bit);
B.Offset += SizeInt;
_StoreMemAutoTSO(Class, Size, B, Value2, OpSize::i8Bit);
_StoreMemAutoTSO(Class, Size, A, Value1, OpSize::i8Bit);
A.Offset += SizeInt;
_StoreMemAutoTSO(Class, Size, A, Value2, OpSize::i8Bit);
}
}
void _StoreMemPairFPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value1, Ref Value2, OpSize Align = OpSize::i8Bit) {
return _StoreMemPairAutoTSO(RegClass::FPR, Size, A, Value1, Value2, Align);
}
Ref Pop(IR::OpSize Size, Ref SP_RMW) {
Ref Value = _AllocateGPR(false);
@@ -35,16 +35,20 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
} else {
LOGMAN_THROW_A_FMT(IsOperandMem(Operand, true), "only memory sources");
AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
AddressMode HighA = A;
HighA.Offset += 16;
if (Operand.IsSIB()) {
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
LOGMAN_THROW_A_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
}
const AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
if (NeedsHigh) {
return _LoadMemPairFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit);
return _LoadMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit);
} else {
return {.Low = _LoadMemFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit)};
return {.Low = _LoadMemAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit)};
}
}
}
@@ -52,8 +56,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
OpDispatchBuilder::RefVSIB
OpDispatchBuilder::AVX128_LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags, bool NeedsHigh) {
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
LOGMAN_THROW_A_FMT((Operand.IsSIB() || Operand.IsSIBRelocation()) && IsVSIB, "Trying to load VSIB for something that isn't the correct "
"type!");
LOGMAN_THROW_A_FMT(Operand.IsSIB() && IsVSIB, "Trying to load VSIB for something that isn't the correct type!");
// VSIB is a very special case which has a ton of encoded data.
// Get it in a format we can reason about.
@@ -65,25 +68,13 @@ OpDispatchBuilder::AVX128_LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tabl
"Base must be a GPR.");
const auto Index_XMM_gpr = Index_gpr - X86State::REG_XMM_0;
OpDispatchBuilder::RefVSIB A {
return {
.Low = AVX128_LoadXMMRegister(Index_XMM_gpr, false),
.High = NeedsHigh ? AVX128_LoadXMMRegister(Index_XMM_gpr, true) : Invalid(),
.BaseAddr = Base_gpr != FEXCore::X86State::REG_INVALID ? LoadGPRRegister(Base_gpr, OpSize::i64Bit, 0, false) : nullptr,
.Displacement = Operand.Data.SIB.Offset,
.Scale = Operand.Data.SIB.Scale,
};
if (Operand.IsSIBRelocation()) {
auto EPOffset = _EntrypointOffset(OpSize::i64Bit, Operand.Data.SIB.Offset);
if (A.BaseAddr) {
A.BaseAddr = Add(OpSize::i64Bit, EPOffset, A.BaseAddr);
} else {
A.BaseAddr = EPOffset;
}
} else {
A.Displacement = static_cast<int32_t>(Operand.Data.SIB.Offset);
}
return A;
}
void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand,
@@ -104,9 +95,9 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
if (Src.High) {
_StoreMemPairFPRAutoTSO(OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
_StoreMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
} else {
_StoreMemFPRAutoTSO(OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
_StoreMemAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
}
}
}
@@ -160,13 +151,13 @@ void OpDispatchBuilder::AVX128_VMOVScalarImpl(OpcodeArgs, IR::OpSize ElementSize
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Result, .High = High});
} else if (Op->Dest.IsGPR()) {
// VMOVSS/SD xmm1, mem32/mem64
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags);
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], ElementSize, Op->Flags);
auto High = LoadZeroVector(OpSize::i128Bit);
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Src, .High = High});
} else {
// VMOVSS/SD mem32/mem64, xmm1
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, ElementSize);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, ElementSize, OpSize::iInvalid);
}
}
@@ -360,7 +351,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
if (Op->Dest.IsGPR()) {
///< MOVNTDQA load non-temporal comes from SSE4.1 and is extended by AVX/AVX2.
RefPair Src {};
Ref SrcAddr = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
Ref SrcAddr = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
Src.Low = _VLoadNonTemporal(OpSize::i128Bit, SrcAddr, 0);
if (Is128Bit) {
@@ -371,7 +362,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
} else {
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit, MemoryAccessType::STREAM);
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
if (Is128Bit) {
// Single store non-temporal for 128-bit operations.
@@ -388,7 +379,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
if (Op->Src[0].IsGPR()) {
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
} else {
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
}
// This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 256bit
@@ -399,7 +390,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
Src.High = ZeroVector;
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
} else {
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
}
}
@@ -408,7 +399,7 @@ void OpDispatchBuilder::AVX128_VMOVLP(OpcodeArgs) {
if (!Op->Dest.IsGPR()) {
///< VMOVLPS/PD mem64, xmm1
StoreResultFPR_WithOpSize(Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
} else if (!Op->Src[1].IsGPR()) {
///< VMOVLPS/PD xmm1, xmm2, mem64
// Bits[63:0] come from Src2[63:0]
@@ -472,7 +463,7 @@ void OpDispatchBuilder::AVX128_VMOVDDUP(OpcodeArgs) {
// 128-bit operation only loads 8-bytes.
// 256-bit operation loads a full 32-bytes.
if (Is128Bit) {
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
} else {
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, true);
}
@@ -567,18 +558,18 @@ void OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstEle
if (Op->Src[1].IsGPR()) {
// If the source is a GPR then convert directly from the GPR.
auto Src2 = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GetGPROpSize(), Op->Flags);
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], GetGPROpSize(), Op->Flags);
Result.Low = _VSToFGPRInsert(OpSize::i128Bit, DstElementSize, SrcSize, Src1.Low, Src2, false);
} else if (SrcSize != DstElementSize) {
// If the source is from memory but the Source size and destination size aren't the same,
// then it is more optimal to load in to a GPR and convert between GPR->FPR.
// ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't.
auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags);
auto Src2 = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags);
Result.Low = _VSToFGPRInsert(DstSize, DstElementSize, SrcSize, Src1.Low, Src2, false);
} else {
// In the case of cvtsi2s{s,d} where the source and destination are the same size,
// then it is more optimal to load in to the FPR register directly and convert there.
auto Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
auto Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
// Always signed
Result.Low = _VSToFVectorInsert(DstSize, DstElementSize, DstElementSize, Src1.Low, Src2, false, false);
}
@@ -598,11 +589,11 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi
if (Op->Src[0].IsGPR()) {
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
} else {
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcElementSize, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcElementSize, Op->Flags);
}
Ref Result = CVTFPR_To_GPRImpl(Op, Src.Low, SrcElementSize, HostRoundingMode);
StoreResultGPR(Op, Result);
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
@@ -645,7 +636,7 @@ void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize) {
if (Op->Src[0].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
} else {
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
}
Comiss(ElementSize, Src1.Low, Src2.Low);
@@ -662,7 +653,7 @@ void OpDispatchBuilder::AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IR
if (Op->Src[1].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
} else {
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
}
// If OpSize == ElementSize then it only does the lower scalar op
@@ -699,7 +690,7 @@ void OpDispatchBuilder::AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSi
if (Op->Src[1].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
} else {
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
}
const uint8_t CompType = Op->Src[2].Literal();
@@ -717,12 +708,12 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
RefPair Result {};
if (Op->Src[0].IsGPR()) {
// Loading from GPR and moving to Vector.
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], GetGPROpSize(), Op->Flags);
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], GetGPROpSize(), Op->Flags);
// zext to 128bit
Result.Low = _VCastFromGPR(OpSize::i128Bit, OpSizeFromSrc(Op), Src);
} else {
// Loading from Memory as a scalar. Zero extend
Result.Low = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Result.Low = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
}
Result.High = LoadZeroVector(OpSize::i128Bit);
@@ -735,11 +726,11 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
auto ElementSize = OpSizeFromDst(Op);
// Extract element from GPR. Zero extending in the process.
Src.Low = _VExtractToGPR(OpSizeFromSrc(Op), ElementSize, Src.Low, 0);
StoreResultGPR(Op, Op->Dest, Src.Low);
StoreResult(GPRClass, Op, Op->Dest, Src.Low, OpSize::iInvalid);
} else {
// Storing first element to memory.
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
_StoreMemFPR(OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
_StoreMem(FPRClass, OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
}
}
}
@@ -767,7 +758,7 @@ void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs, IR::OpSize ElementSize) {
const auto GPRSize = GetGPROpSize();
// Extract already zero extends the result.
Ref Result = _VExtractToGPR(OpSize::i128Bit, OverridenElementSize, Src.Low, Index);
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, GPRSize);
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
return;
}
@@ -788,7 +779,7 @@ void OpDispatchBuilder::AVX128_ExtendVectorElements(OpcodeArgs, IR::OpSize Eleme
const auto SrcSize = OpSizeFromSrc(Op);
const auto LoadSize = Is256Bit ? IR::SizeToOpSize(IR::OpSizeToSize(SrcSize) * 2) : SrcSize;
return LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
return LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
}
};
@@ -877,7 +868,7 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
auto GPRHigh = Mask8Byte(Src.High);
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 2);
}
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid);
}
void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
@@ -906,7 +897,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
Result = _Orlshl(OpSize::i64Bit, Result, ResultHigh, 16);
}
StoreResultGPR(Op, Result);
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, const X86Tables::DecodedOperand& Src1Op,
@@ -919,7 +910,7 @@ void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, con
if (Src2Op.IsGPR()) {
// If the source is a GPR then convert directly from the GPR.
auto Src2 = LoadSourceGPR_WithOpSize(Op, Src2Op, GetGPROpSize(), Op->Flags);
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, GetGPROpSize(), Op->Flags);
Result.Low = _VInsGPR(OpSize::i128Bit, ElementSize, Index, Src1.Low, Src2);
} else {
// If loading from memory then we only load the element size
@@ -1056,7 +1047,7 @@ void OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::O
// Then zero extends the top 128-bit.
const auto SrcSize = Op->Src[1].IsGPR() ? OpSize::i128Bit : SrcElementSize;
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
Ref Result = _VFToFScalarInsert(OpSize::i128Bit, DstElementSize, SrcElementSize, Src1.Low, Src2, false);
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result));
@@ -1085,7 +1076,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize
} else {
// Handle 64-bit memory source.
// In the case of cvtps2pd xmm, m64.
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
}
RefPair Result {};
@@ -1163,7 +1154,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize Sr
// unnecessarily zero extend the vector. Otherwise, if
// memory, then we want to load the element size exactly.
const auto LoadSize = IR::SizeToOpSize(8 * (IR::OpSizeToSize(Size) / 16));
return RefPair {.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags)};
return RefPair {.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags)};
} else {
return AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
}
@@ -1313,7 +1304,7 @@ void OpDispatchBuilder::AVX128_InsertScalarRound(OpcodeArgs, IR::OpSize ElementS
if (Op->Src[1].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
} else {
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
}
// If OpSize == ElementSize then it only does the lower scalar op
@@ -1582,20 +1573,20 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
auto Address = MakeAddress(Op->Dest);
auto Data = AVX128_LoadSource_WithOpSize(Op, DataOp, Op->Flags, !Is128Bit);
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MemOffsetType::SXTX, 1);
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
if (!Is128Bit) {
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MemOffsetType::SXTX, 1);
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
}
} else {
auto Address = MakeAddress(DataOp);
RefPair Result {};
Result.Low = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Address, Invalid(), MemOffsetType::SXTX, 1);
Result.Low = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
if (Is128Bit) {
Result.High = LoadZeroVector(OpSize::i128Bit);
} else {
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MemOffsetType::SXTX, 1);
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
}
@@ -1625,11 +1616,11 @@ void OpDispatchBuilder::AVX128_MASKMOV(OpcodeArgs) {
// RDI source (DS prefix by default)
auto MemDest = MakeSegmentAddress(X86State::REG_RDI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
Ref XMMReg = _LoadMemFPR(Size, MemDest, OpSize::i8Bit);
Ref XMMReg = _LoadMem(FPRClass, Size, MemDest, OpSize::i8Bit);
// If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant
XMMReg = _VBSL(Size, MaskSrc.Low, VectorSrc.Low, XMMReg);
_StoreMemFPR(Size, MemDest, XMMReg, OpSize::i8Bit);
_StoreMem(FPRClass, Size, MemDest, XMMReg, OpSize::i8Bit);
}
void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) {
@@ -1669,7 +1660,7 @@ void OpDispatchBuilder::AVX128_SaveAVXState(Ref MemBase) {
for (uint32_t i = 0; i < NumRegs; i += 2) {
RefPair Pair = LoadContextPair(OpSize::i128Bit, AVXHigh0Index + i);
_StoreMemPairFPR(OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
_StoreMemPair(FPRClass, OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
}
}
@@ -1677,7 +1668,7 @@ void OpDispatchBuilder::AVX128_RestoreAVXState(Ref MemBase) {
const auto NumRegs = Is64BitMode ? 16U : 8U;
for (uint32_t i = 0; i < NumRegs; i += 2) {
auto YMMHRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 576);
auto YMMHRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 576);
AVX128_StoreXMMRegister(i, YMMHRegs.Low, true);
AVX128_StoreXMMRegister(i + 1, YMMHRegs.High, true);
@@ -1969,7 +1960,7 @@ void OpDispatchBuilder::AVX128_VFMAImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx,
}
void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx) {
const OpSize ElementSize = Op->Flags & X86Tables::DecodeFlags::FLAG_OPTION_AVX_W ? OpSize::i64Bit : OpSize::i32Bit;
const auto SrcSize = OpSizeFromSrc(Op);
auto Dest = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, false).Low;
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false).Low;
@@ -1977,13 +1968,13 @@ void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Sr
if (Op->Src[1].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false).Low;
} else {
Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags);
Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
}
Ref Sources[3] = {Dest, Src1, Src2};
DeriveOp(Result_Low, IROp,
_VFMLAScalarInsert(OpSize::i128Bit, ElementSize, Dest, Sources[Src1Idx - 1], Sources[Src2Idx - 1], Sources[AddendIdx - 1]));
_VFMLAScalarInsert(OpSize::i128Bit, SrcSize, Dest, Sources[Src1Idx - 1], Sources[Src2Idx - 1], Sources[AddendIdx - 1]));
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result_Low));
}
@@ -2025,8 +2016,8 @@ void OpDispatchBuilder::AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Sr
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
}
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize,
RefPair Dest, RefPair Mask, RefVSIB VSIB) {
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest,
RefPair Mask, RefVSIB VSIB) {
LOGMAN_THROW_A_FMT(AddrElementSize == OpSize::i32Bit || AddrElementSize == OpSize::i64Bit, "Unknown address element size");
const auto Is128Bit = Size == OpSize::i128Bit;
@@ -2070,13 +2061,10 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, Op
}
}
const auto GPRSize = GetGPROpSize();
auto AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
RefPair Result {};
///< Calculate the low-half.
Result.Low = _VLoadVectorGatherMasked(OpSize::i128Bit, ElementLoadSize, Dest.Low, Mask.Low, BaseAddr, VSIB.Low, VSIB.High,
AddrElementSize, VSIB.Scale, 0, 0, AddrSize);
AddrElementSize, VSIB.Scale, 0, 0);
if (Is128Bit) {
Result.High = LoadZeroVector(OpSize::i128Bit);
@@ -2113,7 +2101,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, Op
///< Calculate the high-half.
auto ResultHigh = _VLoadVectorGatherMasked(OpSize::i128Bit, ElementLoadSize, DestReg, MaskReg, BaseAddr, AddrAddressing.Low,
AddrAddressing.High, AddrElementSize, VSIB.Scale, DataElementOffset, IndexElementOffset, AddrSize);
AddrAddressing.High, AddrElementSize, VSIB.Scale, DataElementOffset, IndexElementOffset);
if (AddrElementSize == OpSize::i64Bit && ElementLoadSize == OpSize::i32Bit) {
// If we only fetched 128-bits worth of data then the upper-result is all zero.
@@ -2126,7 +2114,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, Op
return Result;
}
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(OpcodeArgs, Ref Dest, Ref Mask, RefVSIB VSIB) {
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB) {
///< BaseAddr doesn't need to exist, calculate that here.
Ref BaseAddr = VSIB.BaseAddr;
@@ -2154,11 +2142,8 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(OpcodeArgs,
RefPair Result {};
const auto GPRSize = GetGPROpSize();
auto AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
///< Calculate the low-half.
Result.Low = _VLoadVectorGatherMaskedQPS(OpSize::i128Bit, OpSize::i32Bit, Dest, Mask, BaseAddr, VSIB.Low, VSIB.High, VSIB.Scale, AddrSize);
Result.Low = _VLoadVectorGatherMaskedQPS(OpSize::i128Bit, OpSize::i32Bit, Dest, Mask, BaseAddr, VSIB.Low, VSIB.High, VSIB.Scale);
Result.High = LoadZeroVector(OpSize::i128Bit);
if (VSIB.High == Invalid()) {
// Special case for only loading two floats.
@@ -2217,15 +2202,15 @@ void OpDispatchBuilder::AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize) {
}
///< AddressElementSize is now OpSize::i64Bit
Result = AVX128_VPGatherQPSImpl(Op, Dest.Low, Mask.Low, VSIBLow);
Result = AVX128_VPGatherQPSImpl(Dest.Low, Mask.Low, VSIBLow);
if (NeedsHighAddrBytes) {
auto Res = AVX128_VPGatherQPSImpl(Op, Dest.High, Mask.High, VSIBHigh);
auto Res = AVX128_VPGatherQPSImpl(Dest.High, Mask.High, VSIBHigh);
Result.High = Res.Low;
}
} else if (AddrElementSize == OpSize::i64Bit && ElementLoadSize == OpSize::i32Bit) {
Result = AVX128_VPGatherQPSImpl(Op, Dest.Low, Mask.Low, VSIB);
Result = AVX128_VPGatherQPSImpl(Dest.Low, Mask.Low, VSIB);
} else {
Result = AVX128_VPGatherImpl(Op, Size, ElementLoadSize, AddrElementSize, Dest, Mask, VSIB);
Result = AVX128_VPGatherImpl(Size, ElementLoadSize, AddrElementSize, Dest, Mask, VSIB);
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
@@ -2249,7 +2234,7 @@ void OpDispatchBuilder::AVX128_VCVTPH2PS(OpcodeArgs) {
// In the event that a memory operand is used as the source operand,
// the access width will always be half the size of the destination vector width
// (i.e. 128-bit vector -> 64-bit mem, 256-bit vector -> 128-bit mem)
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
}
RefPair Result {};
@@ -2304,7 +2289,7 @@ void OpDispatchBuilder::AVX128_VCVTPS2PH(OpcodeArgs) {
}
if (!Op->Dest.IsGPR()) {
StoreResultFPR_WithOpSize(Op, Op->Dest, Result.Low, StoreSize);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result.Low, StoreSize, OpSize::iInvalid);
} else {
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
}
@@ -23,8 +23,8 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
@@ -36,7 +36,7 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
@@ -44,15 +44,15 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
@@ -60,8 +60,8 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
auto Src1 = SHADataShuffle(Dest);
@@ -70,7 +70,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
// The result is swizzled differently than expected
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
@@ -79,8 +79,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
return;
}
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result {};
Ref ConstantVector {};
@@ -112,7 +112,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
}
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
@@ -120,12 +120,12 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto Result = _VSha256U0(Dest, Src);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
@@ -133,8 +133,8 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
@@ -142,7 +142,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
auto Result = _VSha256U1(Src1, Src2);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
@@ -150,8 +150,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// Hardcoded to XMM0
auto XMM0 = LoadXMMRegister(0);
@@ -177,7 +177,7 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
auto B = _VSha256H2(EFGH, ABCD, Key);
auto Result = shuffle_abcd(A, B);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
@@ -185,9 +185,9 @@ void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESImc(Src);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
@@ -195,10 +195,10 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
@@ -208,11 +208,11 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESENC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
@@ -220,10 +220,10 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
@@ -233,11 +233,11 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESENCLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
@@ -245,10 +245,10 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
@@ -258,11 +258,11 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESDEC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
@@ -270,10 +270,10 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
@@ -283,15 +283,15 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESDECLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const uint64_t RCON = Op->Src[1].Literal();
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
@@ -305,7 +305,7 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
}
Ref Result = AESKeyGenAssistImpl(Op);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
@@ -313,12 +313,12 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
StoreResultFPR(Op, Res);
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
}
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
@@ -328,12 +328,12 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
}
const auto DstSize = OpSizeFromDst(Op);
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
StoreResultFPR(Op, Res);
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
}
} // namespace FEXCore::IR
@@ -263,7 +263,7 @@ void OpDispatchBuilder::CalculateDeferredFlags() {
Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
// If CF not inverted, we use .cc since the increment happens when the
// condition is false. If CF inverted, invert to use .cs. A bit mindbendy.
return _NZCVSelectIncrement(OpSize, CFInverted ? CondClass::UGE : CondClass::ULT, Src, Src);
return _NZCVSelectIncrement(OpSize, {CFInverted ? COND_UGE : COND_ULT}, Src, Src);
}
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
@@ -290,7 +290,7 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
// TODO: We can fold that second Bfe in (cmp uxth).
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Res, Src2PlusCF);
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Res, Src2PlusCF);
SetNZ_ZeroCV(SrcSize, Res);
SetCFInverted(SelectCFInv);
@@ -324,7 +324,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
Res = Sub(OpSize, Src1, Src2PlusCF);
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Src1, Src2PlusCF);
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Src1, Src2PlusCF);
SetNZ_ZeroCV(SrcSize, Res);
SetCFInverted(SelectCFInv);
@@ -406,7 +406,7 @@ void OpDispatchBuilder::CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High
// If High = SignBit, then sets to nZCv. Else sets to nzcV. Since SF/ZF
// undefined, this does what we need after inverting carry.
auto Zero = _InlineConstant(0);
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
CFInverted = true;
}
@@ -423,7 +423,7 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
// If High = 0, then sets to nZCv. Else sets to nzcV. Since SF/ZF undefined,
// this does what we need.
_CondSubNZCV(Size, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
_CondSubNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
CFInverted = true;
}
@@ -151,9 +151,6 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 4), 4, &OpDispatchBuilder::NOPOp},
// GROUP 17
{OPD(FEXCore::X86Tables::TYPE_GROUP_17, PF_66, 0), 1, &OpDispatchBuilder::Extrq_imm},
// GROUP P
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, true, false, 1>},
@@ -145,7 +145,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
#ifndef _WIN32
// FEX reserved instructions
{0x3E, 1, &OpDispatchBuilder::CallbackReturnOp},
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
#endif
};
@@ -198,8 +198,6 @@ constexpr DispatchTableEntry OpDispatch_SecondaryRepNEModTables[] = {
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
{0x78, 1, &OpDispatchBuilder::Insertq_imm},
{0x79, 1, &OpDispatchBuilder::Insertq},
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i32Bit>},
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
@@ -258,7 +256,6 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>},
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
{0x78, 1, nullptr}, // GROUP 17
{0x79, 1, &OpDispatchBuilder::Extrq},
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>},
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i64Bit>},
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
File diff suppressed because it is too large. Load diff
@@ -17,6 +17,7 @@ $end_info$
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/FPState.h>
#include <cmath>
#include <stddef.h>
#include <stdint.h>
@@ -27,12 +28,10 @@ class OrderedNode;
Ref OpDispatchBuilder::GetX87Top() {
// Yes, we are storing 3 bits in a single flag register.
// Deal with it
return _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
}
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
_StackForceSlow(); // Invalidate x87 FTW register cache
// For the output, we want a 1-bit for each pair not equal to 11 (Empty).
static_assert(static_cast<uint8_t>(FPState::X87Tag::Empty) == 0b11);
@@ -51,22 +50,24 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4);
// ...and that's it. StoreContext implicitly does the final masking.
_StoreContextGPR(OpSize::i8Bit, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, FTW);
}
void OpDispatchBuilder::SetX87Top(Ref Value) {
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
}
// Float LoaD operation with memory operand
void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) {
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
Ref ConvertedData = Data;
// Convert to 80bit float
if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
ConvertedData = _F80CVTTo(Data, Width);
ConvertedData = _F80CVTTo(Data, ReadWidth);
}
_PushStack(ConvertedData, Data, Width);
_PushStack(ConvertedData, Data, ReadWidth, true);
}
// Float LoaD operation with memory operand
@@ -76,27 +77,27 @@ void OpDispatchBuilder::FLDFromStack(OpcodeArgs) {
void OpDispatchBuilder::FBLD(OpcodeArgs) {
// Read from memory
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
Ref ConvertedData = _F80BCDLoad(Data);
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
_PushStack(ConvertedData, Data, OpSize::i128Bit, true);
}
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
Ref converted = _F80BCDStore(_ReadStackValue(0));
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
_PopStackDestroy();
}
void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant K) {
// Update TOP
Ref Data = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, K);
_PushStack(Data, Data, OpSize::f80Bit);
_PushStack(Data, Data, OpSize::i128Bit, true);
}
void OpDispatchBuilder::FILD(OpcodeArgs) {
const auto ReadWidth = OpSizeFromSrc(Op);
// Read from memory
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
// Sign extend to 64bits
if (ReadWidth != OpSize::i64Bit) {
@@ -109,28 +110,27 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
// Extract sign and make integer absolute
auto zero = Constant(0);
_SubNZCV(OpSize::i64Bit, Data, zero);
auto sign = _NZCVSelect(OpSize::i64Bit, CondClass::SLT, Constant(0x8000), zero);
auto absolute = _Neg(OpSize::i64Bit, Data, CondClass::MI);
auto sign = _NZCVSelect(OpSize::i64Bit, CondClassType {COND_SLT}, Constant(0x8000), zero);
auto absolute = _Neg(OpSize::i64Bit, Data, CondClassType {COND_MI});
// left justify the absolute integer
auto shift = Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
auto shifted = _Lshl(OpSize::i64Bit, absolute, shift);
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
auto zeroed_exponent = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, absolute, zero, zero, adjusted_exponent);
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
Ref ConvertedData = _VLoadTwoGPRs(shifted, upper);
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
_PushStack(ConvertedData, Data, ReadWidth, false);
}
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
LOGMAN_THROW_A_FMT(Width == OpSize::i32Bit || Width == OpSize::i64Bit || Width == OpSize::f80Bit, "Invalid store width for FST");
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::f80Bit;
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, false, false, Width);
_StoreStackMem(SourceSize, Width, A.Base, A.Index, OpSize::iInvalid, A.IndexType, A.IndexScale);
_StoreStackMem(SourceSize, Width, A.Base, A.Index, OpSize::iInvalid, A.IndexType, A.IndexScale, /*Float=*/true);
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
_PopStackDestroy();
@@ -164,12 +164,12 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
// Check for NaN/Infinity: exponent = 0x7fff
SaveNZCV();
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff));
Ref IsSpecial = _NZCVSelect01(CondClass::EQ);
Ref IsSpecial = _NZCVSelect01({COND_EQ});
// For overflow detection, check if exponent indicates a value >= 2^15
// Biased exponent for 2^15 is 0x3fff + 15 = 0x400e
SubWithFlags(OpSize::i64Bit, Exponent, 0x400e);
Ref IsOverflow = _NZCVSelect01(CondClass::UGE);
Ref IsOverflow = _NZCVSelect01({COND_UGE});
// Set Invalid Operation flag if overflow or special value
Ref InvalidFlag = _Or(OpSize::i64Bit, IsSpecial, IsOverflow);
@@ -178,7 +178,7 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
Data = _F80CVTInt(Size, Data, Truncate);
StoreResultGPR_WithOpSize(Op, Op->Dest, Data, Size, OpSize::i8Bit);
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, OpSize::i8Bit);
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
_PopStackDestroy();
@@ -204,10 +204,10 @@ void OpDispatchBuilder::FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
// We have one memory argument
Ref Arg {};
if (Integer) {
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
Arg = _F80CVTToInt(Arg, Width);
} else {
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Arg = _F80CVTTo(Arg, Width);
}
@@ -234,10 +234,10 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
// We have one memory argument
Ref arg {};
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
arg = _F80CVTToInt(arg, Width);
} else {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _F80CVTTo(arg, Width);
}
@@ -271,10 +271,10 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
// We have one memory argument
Ref arg {};
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
arg = _F80CVTToInt(arg, Width);
} else {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _F80CVTTo(arg, Width);
}
@@ -312,10 +312,10 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
// We have one memory argument
Ref Arg {};
if (Integer) {
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
Arg = _F80CVTToInt(Arg, Width);
} else {
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Arg = _F80CVTTo(Arg, Width);
}
@@ -338,7 +338,7 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() {
// bytes, we use the well-known bit twiddling algorithm:
//
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
Ref X = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref X = LoadContext(AbridgedFTWIndex);
X = _Orlshl(OpSize::i32Bit, X, X, 4);
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f));
X = _Orlshl(OpSize::i32Bit, X, X, 2);
@@ -379,41 +379,41 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
_SyncStackToSlow();
const auto Size = OpSizeFromSrc(Op);
Ref Mem = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
Mem = AppendSegmentOffset(Mem, Op->Flags);
{
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMemGPR(Size, Mem, FCW, Size);
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMem(GPRClass, Size, Mem, FCW, Size);
}
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
auto ZeroConst = Constant(0);
{
// FTW
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
}
{
// Instruction Offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
}
{
// Instruction CS selector (+ Opcode)
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
}
{
// Data pointer offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
}
{
// Data pointer selector
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
}
}
@@ -439,20 +439,20 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
_StackForceSlow();
const auto Size = OpSizeFromSrc(Op);
Ref Mem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
Ref Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
Mem = AppendSegmentOffset(Mem, Op->Flags);
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 1);
auto NewFSW = _LoadMemGPR(Size, MemLocation, Size);
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
ReconstructX87StateFromFSW_Helper(NewFSW);
{
// FTW
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 2);
SetX87FTW(_LoadMemGPR(Size, MemLocation, Size));
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
}
}
@@ -481,61 +481,61 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
Ref Top = GetX87Top();
{
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMemGPR(Size, Mem, FCW, Size);
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMem(GPRClass, Size, Mem, FCW, Size);
}
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
auto ZeroConst = Constant(0);
{
// FTW
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
}
{
// Instruction Offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
}
{
// Instruction CS selector (+ Opcode)
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
}
{
// Data pointer offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
}
{
// Data pointer selector
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
}
auto SevenConst = Constant(7);
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (int i = 0; i < 7; ++i) {
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
if (ReducedPrecisionMode) {
data = _F80CVTTo(data, OpSize::i64Bit);
}
_StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreMem(FPRClass, OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
}
// The final st(7) needs a bit of special handling here
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
if (ReducedPrecisionMode) {
data = _F80CVTTo(data, OpSize::i64Bit);
}
// ST7 broken in to two parts
// Lower 64bits [63:0]
// upper 16 bits [79:64]
_StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreMem(FPRClass, OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4);
_StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreMem(FPRClass, OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
// reset to default
FNINIT(Op);
@@ -546,8 +546,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
const auto Size = OpSizeFromSrc(Op);
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
if (ReducedPrecisionMode) {
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
@@ -559,11 +559,11 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
_SetRoundingMode(roundingMode, false, roundingMode);
}
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1);
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1);
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
{
// FTW
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
}
auto SevenConst = Constant(7);
@@ -572,14 +572,14 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
Ref Mask = _VLoadTwoGPRs(low, high);
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (int i = 0; i < 7; ++i) {
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
// Mask off the top bits
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
if (ReducedPrecisionMode) {
// Convert to double precision
Reg = _F80CVT(OpSize::i64Bit, Reg);
}
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
}
@@ -588,19 +588,20 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
// ST7 broken in to two parts
// Lower 64bits [63:0]
// upper 16 bits [79:64]
Ref Reg = _LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Ref RegHigh =
_LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
if (ReducedPrecisionMode) {
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
}
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
}
// Load / Store Control Word
void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
StoreResultGPR(Op, FCW);
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
StoreResult(GPRClass, Op, FCW, OpSize::iInvalid);
}
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
@@ -608,8 +609,8 @@ void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
// to switch for now to slow mode whenever these are manually changed.
// Remove the next line and try DF_04.asm in fast path.
_StackForceSlow();
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
}
void OpDispatchBuilder::FXCH(OpcodeArgs) {
@@ -645,10 +646,10 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
// Memory arg
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
b = _F80CVTToInt(arg, Width);
} else {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
b = _F80CVTTo(arg, Width);
}
} else {
@@ -764,7 +765,7 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
Ref TopValue = _SyncStackToSlow();
Ref StatusWord = ReconstructFSW_Helper(TopValue);
StoreResultGPR(Op, StatusWord);
StoreResult(GPRClass, Op, StatusWord, OpSize::iInvalid);
}
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
@@ -773,8 +774,6 @@ void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
}
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
_SyncStackToSlow(); // Invalidate x87 register caches
auto Zero = Constant(0);
if (ReducedPrecisionMode) {
@@ -783,12 +782,12 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
// Init FCW to 0x037F
auto NewFCW = Constant(0x037F);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
// Set top to zero
SetX87Top(Zero);
// Tags all get marked as invalid
_StoreContextGPR(OpSize::i8Bit, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, Zero);
// Reinits the simulated stack
_InitStack();
@@ -861,7 +860,7 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
auto TopValid = _StackValidTag(0);
// In the case of top being invalid then C3:C2:C0 is 0b101
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
auto C3 = Select01(OpSize::i32Bit, CondClassType {COND_NEQ}, TopValid, Constant(1));
auto C2 = TopValid;
auto C0 = C3; // Mirror C3 until something other than zero is supported
@@ -876,8 +875,8 @@ void OpDispatchBuilder::X87FXTRACT(OpcodeArgs) {
_PopStackDestroy();
auto Exp = _F80XTRACT_EXP(Top);
auto Sig = _F80XTRACT_SIG(Top);
_PushStack(Exp, Invalid(), OpSize::iInvalid);
_PushStack(Sig, Invalid(), OpSize::iInvalid);
_PushStack(Exp, Exp, OpSize::f80Bit, true);
_PushStack(Sig, Sig, OpSize::f80Bit, true);
}
} // namespace FEXCore::IR
@@ -29,37 +29,38 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
const auto Size = OpSizeFromSrc(Op);
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
_SetRoundingMode(roundingMode, false, roundingMode);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1);
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MEM_OFFSET_SXTX, 1);
ReconstructX87StateFromFSW_Helper(NewFSW);
{
// FTW
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
}
}
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
_StackForceSlow();
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
_SetRoundingMode(roundingMode, false, roundingMode);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
}
// F64 ops
// Float load op with memory operand
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
// Convert to 64bit float
Ref ConvertedData = Data;
if (Width == OpSize::i32Bit) {
@@ -67,39 +68,39 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
} else if (Width == OpSize::f80Bit) {
ConvertedData = _F80CVT(OpSize::i64Bit, Data);
}
_PushStack(ConvertedData, Data, Width);
_PushStack(ConvertedData, Data, ReadWidth, true);
}
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
// Read from memory
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i128Bit, Op->Flags);
Ref ConvertedData = _F80BCDLoad(Data);
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
}
void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
Ref converted = _F80CVTTo(_ReadStackValue(0), OpSize::i64Bit);
converted = _F80BCDStore(converted);
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
_PopStackDestroy();
}
void OpDispatchBuilder::FLDF64_Const(OpcodeArgs, uint64_t Num) {
auto Data = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(Num));
_PushStack(Data, Data, OpSize::i64Bit);
_PushStack(Data, Data, OpSize::i64Bit, true);
}
void OpDispatchBuilder::FILDF64(OpcodeArgs) {
const auto ReadWidth = OpSizeFromSrc(Op);
// Read from memory
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
if (ReadWidth == OpSize::i16Bit) {
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
}
auto ConvertedData = _Float_FromGPR_S(OpSize::i64Bit, ReadWidth == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, Data);
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
_PushStack(ConvertedData, Data, ReadWidth, false);
}
void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
@@ -111,7 +112,7 @@ void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
} else {
data = _Float_ToGPR_S(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data);
}
StoreResultGPR_WithOpSize(Op, Op->Dest, data, Size, OpSize::i8Bit);
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, OpSize::i8Bit);
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
_PopStackDestroy();
@@ -137,16 +138,16 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
Ref arg {};
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
} else if (Width == OpSize::i32Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
} else if (Width == OpSize::i64Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
} else {
FEX_UNREACHABLE;
}
@@ -175,16 +176,16 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
Ref arg {};
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
} else if (Width == OpSize::i32Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
} else if (Width == OpSize::i64Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
} else {
FEX_UNREACHABLE;
}
@@ -227,16 +228,16 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
if (Integer) {
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
Arg = _Sbfe(OpSize::i64Bit, 16, 0, Arg);
}
Arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Arg);
} else if (Width == OpSize::i32Bit) {
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Arg);
} else if (Width == OpSize::i64Bit) {
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
}
} else {
FEX_UNREACHABLE;
@@ -284,16 +285,16 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
} else if (Width == OpSize::i32Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
} else if (Width == OpSize::i64Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
}
} else {
FEX_UNREACHABLE;
@@ -331,16 +332,16 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
} else if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
// Memory arg
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
} else if (Width == OpSize::i32Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
} else if (Width == OpSize::i64Bit) {
b = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
}
} else {
FEX_UNREACHABLE;
@@ -392,11 +393,11 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
SaveNZCV();
_TestNZ(OpSize::i64Bit, Gpr, Constant(0x7fff'ffff'ffff'ffffUL));
Ref Sig = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, SigZV, SigNZV);
Ref Exp = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, ExpZV, ExpNZV);
Ref Sig = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, SigZV, SigNZV);
Ref Exp = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, ExpZV, ExpNZV);
_PopStackDestroy();
_PushStack(Exp, Invalid(), OpSize::iInvalid);
_PushStack(Sig, Invalid(), OpSize::iInvalid);
_PushStack(Exp, Exp, OpSize::i64Bit, true);
_PushStack(Sig, Sig, OpSize::i64Bit, true);
}
} // namespace FEXCore::IR
@@ -0,0 +1,87 @@
// SPDX-License-Identifier: MIT
/*
$info$
tags: glue|x86-guest-code
desc: Guest-side assembly helpers used by the backends
$end_info$
*/
#include "Interface/Core/X86HelperGen.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <cstdint>
#include <cstring>
namespace FEXCore {
constexpr size_t CODE_SIZE = 0x1000;
X86GeneratedCode::X86GeneratedCode() {
#ifdef _WIN32
// No need to allocate anything in this config.
#else
// Allocate a page for our emulated guest
CodePtr = AllocateGuestCodeSpace(CODE_SIZE);
constexpr std::array<uint8_t, 2> SignalReturnCode = {
0x0F, 0x37, // CALLBACKRET FEX Instruction
};
CallbackReturn = reinterpret_cast<uint64_t>(CodePtr);
memcpy(reinterpret_cast<void*>(CallbackReturn), SignalReturnCode.data(), SignalReturnCode.size());
mprotect(CodePtr, CODE_SIZE, PROT_READ);
#endif
}
X86GeneratedCode::~X86GeneratedCode() {
#ifndef _WIN32
FEXCore::Allocator::VirtualFree(CodePtr, CODE_SIZE);
#endif
}
void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
#ifndef _WIN32
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
if (Is64BitMode()) {
// 64bit mode can have its sigret handler anywhere
return FEXCore::Allocator::VirtualAlloc(Size);
}
// First 64bit page
constexpr uintptr_t LOCATION_MAX = 0x1'0000'0000;
// 32bit mode
// We need to have the sigret handler in the lower 32bits of memory space
// Scan top down and try to allocate a location
for (size_t Location = 0xFFFF'E000; Location != 0x0; Location -= 0x1000) {
void* Ptr = ::mmap(reinterpret_cast<void*>(Location), Size, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
if (Ptr != MAP_FAILED && reinterpret_cast<uintptr_t>(Ptr) >= LOCATION_MAX) {
// Failed to map in the lower 32bits
// Try again
// Can happen in the case that host kernel ignores MAP_FIXED_NOREPLACE
::munmap(Ptr, Size);
continue;
}
if (Ptr != MAP_FAILED) {
return Ptr;
}
}
// Can't do anything about this
// Here's hoping the application doesn't use signals
return MAP_FAILED;
#else
return nullptr;
#endif
}
} // namespace FEXCore
@@ -0,0 +1,25 @@
// SPDX-License-Identifier: MIT
/*
$info$
tags: glue|x86-guest-code
$end_info$
*/
#pragma once
#include <stddef.h>
#include <stdint.h>
namespace FEXCore {
class X86GeneratedCode final {
public:
X86GeneratedCode();
~X86GeneratedCode();
uint64_t CallbackReturn {};
private:
void* CodePtr {};
void* AllocateGuestCodeSpace(size_t Size);
};
} // namespace FEXCore
Loaded 100 of 882 files, more files were not shown because too many files have changed in this diff. Show more