Compare commits

..
6 Commits
Author SHA1 Message Date
Ryan Houdek c094dc238e Docs: Update for release FEX-2509.1 2025-09-15 18:33:36 -07:00
Billy Laws ceaf38e996 Dispatcher: Fix FABI_F32_I16_F80_PTR argument size
This takes an f80 as input and returns an f32. A copy-paste error had
this truncating the input float if !TMP_ABIARGS.
2025-09-15 18:31:55 -07:00
Billy Laws 85e9e255a5 unittests: Add test for x87 mode switches wrongly flushing NZCV 2025-09-15 18:31:50 -07:00
Billy Laws a545865ab7 OpcodeDispatcher: Only flush MMX registers on MMX -> x87 transitions
Flushing other regs is not necessary, and breaks any ConvertNZCVToX87 use
which relies previously saved NZCV values as the flag-setting NZCV op after
the save could trigger a flush of NZCV.
2025-09-15 18:31:44 -07:00
Billy Laws aa8e8f2cb0 OpcodeDispatcher: Don't assert on invalid ALU op encoding 2025-09-15 18:31:37 -07:00
Billy Laws d3a8701e1a WOW64: Fix CsSeg initialization 2025-09-15 18:31:31 -07:00
296 changed files with 35015 additions and 61690 deletions

No files matched your search

-3
View File
@@ -7,6 +7,3 @@ FEXCore/Source/Interface/Core/X86Tables/*
# Inline headers with list-like content that can't be processed individually
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
# Include files in unittests
unittests/*ASM/Includes/*.inc
-2
View File
@@ -20,5 +20,3 @@
# Whole-tree reformat with clang-format-19
5267cde60e7642852d18f20ae8568643bb5293d5
# Minor reformat with clang-format-19
9fdd96af61c969cb5732471223f00eda64b7a069
+1
View File
@@ -34,6 +34,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -41,6 +41,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -34,6 +34,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -33,6 +33,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+2 -1
View File
@@ -48,6 +48,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
@@ -77,7 +78,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
+1
View File
@@ -35,6 +35,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+2 -2
View File
@@ -46,12 +46,12 @@ jobs:
- name: Configure CMake arm64ec
shell: bash
working-directory: ${{runner.workspace}}/build_arm64ec
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
- name: Configure CMake wow64
shell: bash
working-directory: ${{runner.workspace}}/build_wow64
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
- name: Build arm64ec
working-directory: ${{runner.workspace}}/build_arm64ec
+55 -13
View File
@@ -4,6 +4,7 @@ project(FEX C CXX ASM)
INCLUDE (CheckIncludeFiles)
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
option(BUILD_THUNKS "Build thunks" FALSE)
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
@@ -303,8 +304,7 @@ set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-poin
include_directories(External/robin-map/include/)
include(CTest)
if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
add_subdirectory(External/vixl/)
include_directories(SYSTEM External/vixl/src/)
endif()
@@ -319,7 +319,7 @@ if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
set(BUILD_SHARED_LIBS OFF)
@@ -335,7 +335,7 @@ endif()
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
if (BUILD_TESTING)
if (BUILD_TESTS)
find_package(Catch2 3 QUIET)
if (NOT Catch2_FOUND)
add_subdirectory(External/Catch2/)
@@ -345,9 +345,6 @@ if (BUILD_TESTING)
endif()
include(Catch)
else ()
# Override any previously generated test list to avoid running stale test binaries
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
endif()
find_package(fmt QUIET)
@@ -458,8 +455,13 @@ endif()
add_compile_options(-Wall)
if (BUILD_TESTING)
include(CTest)
if (BUILD_TESTS)
message(STATUS "Unit tests are enabled")
if (NOT BUILD_TESTING)
# CMake checks this variable before generating CTestTestfile.cmake
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
endif()
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
if (TEST_JOB_COUNT)
@@ -490,11 +492,10 @@ file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.js
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/)
endforeach()
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
@@ -555,7 +556,6 @@ if (BUILD_THUNKS)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)"
DEPENDS guest-libs
COMPONENT Runtime
)
install(
@@ -565,7 +565,6 @@ if (BUILD_THUNKS)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)"
DEPENDS guest-libs-32
COMPONENT Runtime
)
add_custom_target(uninstall_guest-libs
@@ -607,3 +606,46 @@ if (OVERRIDE_VERSION STREQUAL "detect")
else()
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
endif()
# Parse the version here
# Change something like `FEX-2106.1-76-<hash>` in to a list
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
# Extract the `2106.1` element
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
# Change `2106.1` in to a list
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
# Calculate list size
list(LENGTH DESCRIBE_LIST LIST_SIZE)
# Pull out the major version
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
# Minor version only exists if there is a .1 at the end
# eg: 2106 versus 2106.1
if (LIST_SIZE GREATER 1)
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
endif()
# Package creation
set (CPACK_GENERATOR "DEB")
set (CPACK_PACKAGE_NAME fex-emu)
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/Description.txt")
# Debian defines
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/triggers")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
# binfmt_misc conflicts with qemu-user-static
# We also only install binfmt_misc on aarch64 hosts
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
endif()
include (CPack)
+2 -1
View File
@@ -2244,7 +2244,8 @@ public:
template<IsQOrDRegister T>
void movi(SubRegSize size, T rd, uint64_t Imm, uint16_t Shift = 0) {
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit ||
size == SubRegSize::i64Bit,
"Unsupported movi size");
uint32_t cmode;
+3 -3
View File
@@ -5125,7 +5125,7 @@ private:
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
[[nodiscard]]
static bool IsValidFPValueForImm8(T value) {
const uint64_t bits = std::bit_cast<FloatToEquivalentUInt<T>>(value);
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
static constexpr std::array mantissa_masks {
@@ -5171,7 +5171,7 @@ protected:
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
#endif
const auto bits = std::bit_cast<uint32_t>(value);
const auto bits = FEXCore::BitCast<uint32_t>(value);
const auto sign = (bits & 0x80000000) >> 24;
const auto expb2 = (bits & 0x20000000) >> 23;
const auto b5_to_0 = (bits >> 19) & 0x3F;
@@ -5184,7 +5184,7 @@ protected:
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
#endif
const auto bits = std::bit_cast<uint64_t>(value);
const auto bits = FEXCore::BitCast<uint64_t>(value);
const auto sign = (bits & 0x80000000'00000000) >> 56;
const auto expb2 = (bits & 0x20000000'00000000) >> 55;
const auto b5_to_0 = (bits >> 48) & 0x3F;
+2 -4
View File
@@ -4,8 +4,7 @@ file(GLOB GEN_CONFIG_SOURCES CONFIGURE_DEPENDS *.json.in)
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
# Any configuration file json file that needs to be generated
@@ -22,6 +21,5 @@ foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
# Then install the configured json
install(
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
+3
View File
@@ -0,0 +1,3 @@
x86 and x86-64 Linux emulator
FEX allows you to run x86 applications on ARM64 Linux devices. It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
+18
View File
@@ -0,0 +1,18 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Setup binfmt_misc
update-binfmts --import FEX-x86
update-binfmts --import FEX-x86_64
}
# Install FEXInterpreter hardlink
# Needs to be done before setting up binfmt_misc
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
+17
View File
@@ -0,0 +1,17 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Uninstall
update-binfmts --unimport FEX-x86
update-binfmts --unimport FEX-x86_64
}
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
# Remove FEXInterpreter hardlink
unlink /usr/bin/FEXInterpreter
+1
View File
@@ -0,0 +1 @@
activate-noawait ldconfig
+1 -1
View File
@@ -14,7 +14,7 @@ RUN mkdir build
ARG CC=clang-13
ARG CXX=clang++-13
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTING=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTS=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN ninja
WORKDIR /FEX/build
+2 -4
View File
@@ -10,8 +10,7 @@ function(GenBinFmt Name)
# Then install the configured binfmt
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/
COMPONENT Runtime)
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
endfunction()
if (NOT USE_LEGACY_BINFMTMISC)
@@ -20,8 +19,7 @@ if (NOT USE_LEGACY_BINFMTMISC)
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/
COMPONENT Runtime)
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
else()
GenBinFmt(FEX-x86.in)
GenBinFmt(FEX-x86_64.in)
+1 -1
View File
@@ -1 +1 @@
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+1 -1
View File
@@ -1 +1 @@
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+3 -3
View File
@@ -2,8 +2,8 @@
let
toolchain = pkgs.fetchzip {
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250920/llvm-mingw-20250920-ucrt-ubuntu-22.04-aarch64.tar.xz";
sha256 = "sha256-LaojKjC8KzY+soW5u6eoDoXE3qtYk9Ejr7M3enTqRAE=";
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250305/llvm-mingw-20250305-ucrt-ubuntu-20.04-aarch64.tar.xz";
sha256 = "sha256-cA03/ab9O61eO9+S2JzIXD4V0HzTXK5/AYyxW2d73Po=";
};
cmakeToolchainFile = pkgs.substitute {
@@ -45,7 +45,7 @@ pkgs.mkShell {
fi
'';
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False
FEX_CMAKE_TOOLCHAIN_ARM64EC = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_CMAKE_TOOLCHAIN_WOW64 = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_MESON_CROSSFILE = "--cross-file ${mesonCrossFile}";
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -14,4 +14,4 @@ fi
rm -rf unittests/FEXLinuxTests
set -o xtrace
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTING=ON -DBUILD_FEX_LINUX_TESTS=ON
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTS=ON -DBUILD_FEX_LINUX_TESTS=ON
+1 -1
+1 -1
+1 -1
View File
@@ -78,6 +78,6 @@ install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
DESTINATION include
COMPONENT Development)
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
+165 -6
View File
@@ -118,6 +118,41 @@ def print_man_env_option(name, desc, default, no_json_key):
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
def print_man_options(options):
output_man.write(".Sh OPTIONS\n")
output_man.write(".Bl -tag -width -indent\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
default = op_vals["Default"]
value_type = op_vals["Type"]
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = op_vals["TextDefault"]
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
# Wrap the string argument in quotes
default = "'" + default + "'"
print_man_option(
short,
long,
op_vals["Desc"],
default
)
if (value_type == "strenum"):
Enums = op_vals["Enums"]
output_man.write("\\fBAvailable Options:\\fR\n")
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
output_man.write("\n.sp\n")
output_man.write(".El\n")
def print_man_environment(options):
output_man.write(".Sh ENVIRONMENT\n")
output_man.write(".Bl -tag -width -indent\n")
@@ -159,7 +194,7 @@ def print_man_environment_tail():
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
"This will override the full path",
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -173,7 +208,7 @@ def print_man_environment_tail():
"One must be careful with this option as it will override any applications that load with execve as well"
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -192,8 +227,8 @@ def print_man_environment_tail():
"PORTABLE",
[
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
"For FEX on Linux:",
"These files are instead read from <FEXPath>/fex-emu/ by default.",
"For FEXInterpreter on Linux:",
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
"For Arm64ec/Wow64 WINE builds:",
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
@@ -205,12 +240,20 @@ def print_man_header():
.Dt FEX
.Os Linux
.Sh NAME
.Nm FEX
.Nm FEXLoader
.Nm FEXInterpreter
.Nm FEXBash
.Nd Fast x86-64 and x86 emulation.
.Sh SYNOPSIS
.Nm
.Ar <args> ...
.Op options
.Op Ar --
.Ar Application
<args> ...
.Pp
.Nm FEXInterpreter
.Ar Application
<args> ...
.Pp
.Nm FEXBash
.Ar <args> ...
@@ -318,6 +361,82 @@ def print_config_option(type, group_name, json_name, default_value, short, choic
output_argloader.write("\n");
def print_argloader_options(options):
output_argloader.write("#ifdef BEFORE_PARSE\n")
output_argloader.write("#undef BEFORE_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
default = op_vals["Default"]
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
# Wrap the string argument in quotes
default = "\"" + default + "\""
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = "\"" + op_vals["TextDefault"] + "\""
short = None
choices = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if ("Choices" in op_vals):
choices = op_vals["Choices"]
print_config_option(
op_vals["Type"],
op_group,
op_key,
default,
short,
choices,
op_vals["Desc"])
output_argloader.write("\n")
output_argloader.write("#endif\n")
def print_parse_argloader_options(options):
output_argloader.write("#ifdef AFTER_PARSE\n")
output_argloader.write("#undef AFTER_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
output_argloader.write("if (Options.is_set_by_user(\"{0}\")) {{\n".format(op_key))
value_type = op_vals["Type"]
NeedsString = False
conversion_func = "fextl::fmt::format(\"{}\", "
if ("ArgumentHandler" in op_vals):
NeedsString = True
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
if (value_type == "str"):
NeedsString = True
conversion_func = "std::move("
if (value_type == "bool"):
# boolean values need a decimal specifier. Otherwise fmt prints strings.
conversion_func = "fextl::fmt::format(\"{:d}\", "
if (value_type == "strenum"):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key))
elif (value_type == "strarray"):
# these need a bit more help
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
output_argloader.write("\t}\n")
else:
if (NeedsString):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
else:
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_envloader_options(options):
output_argloader.write("#ifdef ENVLOADER\n")
output_argloader.write("#undef ENVLOADER\n")
@@ -398,6 +517,41 @@ def print_parse_enum_options(options):
output_argloader.write("#endif\n")
def check_for_duplicate_options(options):
short_map = []
long_map = []
# Spin through all the items and see if we have a duplicate option
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
long_invert = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if (op_vals["Type"] == "bool"):
long_invert = "no-" + long
# Check for short key duplication
if (short != None):
if (short in short_map):
raise Exception("Short config '{0}' for option '{1}' has duplicate entry!".format(short, op_key))
else:
short_map.append(short)
# Check for long key duplication
if (long in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long))
else:
long_map.append(long)
# Check for long key duplication
if (long_invert != None):
if (long_invert in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long_invert))
else:
long_map.append(long_invert)
if (len(sys.argv) < 5):
sys.exit()
@@ -414,6 +568,8 @@ json_object = json.loads(json_text)
options = json_object["Options"]
unnamed_options = json_object["UnnamedOptions"]
check_for_duplicate_options(options)
# Generate config include file
output_file = open(output_filename, "w")
print_header()
@@ -425,6 +581,7 @@ output_file.close()
# Generate man file
output_man = open(output_man_page, "w")
print_man_header()
print_man_options(options)
print_man_environment(options)
print_man_tail()
@@ -432,6 +589,8 @@ output_man.close()
# Generate argument loader code
output_argloader = open(output_argumentloader_filename, "w")
print_argloader_options(options);
print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
+59 -53
View File
@@ -58,10 +58,10 @@ class OpDefinition:
JITDispatch: bool
JITDispatchOverride: str
TiedSource: int
Inline: list[str]
Arguments: list[OpArgument]
EmitValidation: list[str]
Desc: list[str]
Inline: list
Arguments: list
EmitValidation: list
Desc: list
def __init__(self):
self.Name = None
@@ -92,14 +92,19 @@ class OpDefinition:
attrs = vars(self)
print(", ".join("%s: %s" % item for item in attrs.items()))
IRTypesToCXX: dict[str, IRType] = {}
CXXTypeToIR: dict[str, IRType] = {}
IROps: list[OpDefinition] = []
IRTypesToCXX = {}
CXXTypeToIR = {}
IROps = []
IROpNameSet: set[str] = set()
IROpNameMap = {}
def is_ssa_type(op_type: str):
return op_type in {"SSA", "GPR", "GPRPair", "FPR"}
def is_ssa_type(type):
if (type == "SSA" or
type == "GPR" or
type == "GPRPair" or
type == "FPR"):
return True
return False
def parse_irtypes(irtypes):
for op_key, op_val in irtypes.items():
@@ -214,8 +219,11 @@ def parse_ops(ops):
OpArg.DefaultInitializer = DefaultInit[1][:-1]
# If SSA type then we can generate validation for this op
if OpArg.IsSSA and OpArg.Type in {"GPR", "GPRPair", "FPR"}:
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == RegClass::Invalid || WalkFindRegClass({ArgName}) == RegClass::{OpArg.Type}")
if (OpArg.IsSSA and
(OpArg.Type == "GPR" or
OpArg.Type == "GPRPair" or
OpArg.Type == "FPR")):
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
OpArg.Name = ArgName
OpArg.NameWithPrefix = NameWithPrefix
@@ -288,28 +296,21 @@ def parse_ops(ops):
#OpDef.print()
# Error on duplicate op
if OpDef.Name in IROpNameSet:
if OpDef.Name in IROpNameMap:
ExitError("Duplicate Op defined! {}".format(OpDef.Name))
IROps.append(OpDef)
IROpNameSet.add(OpDef.Name)
IROpNameMap[OpDef.Name] = 1
# Print out enum values
def print_enums(enums):
def print_enums():
output_file.write("#ifdef IROP_ENUM\n")
output_file.write("enum IROps : uint16_t {\n")
for op in IROps:
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
output_file.write("};\n")
for name, members in enums.items():
output_file.write(f"enum {name} {{\n")
for member in members:
if member:
output_file.write(f"\t{member}\n")
else:
output_file.write("\n")
output_file.write("};\n\n")
output_file.write("};\n")
output_file.write("#undef IROP_ENUM\n")
output_file.write("#endif\n\n")
@@ -407,7 +408,7 @@ def print_ir_sizes():
[[nodiscard, gnu::const]] std::string_view const& GetName(IROps Op);
[[nodiscard, gnu::const]] uint8_t GetArgs(IROps Op);
[[nodiscard, gnu::const]] uint8_t GetRAArgs(IROps Op);
[[nodiscard, gnu::const]] FEXCore::IR::RegClass GetRegClass(IROps Op);
[[nodiscard, gnu::const]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
[[nodiscard, gnu::const]] bool HasSideEffects(IROps Op);
[[nodiscard, gnu::const]] bool ImplicitFlagClobber(IROps Op);
[[nodiscard, gnu::const]] bool GetHasDest(IROps Op);
@@ -421,29 +422,30 @@ def print_ir_sizes():
def print_ir_reg_classes():
output_file.write("#ifdef IROP_REG_CLASSES_IMPL\n")
output_file.write("constexpr std::array<FEXCore::IR::RegClass, IROps::OP_LAST + 1> IRRegClasses = {\n")
output_file.write("constexpr std::array<FEXCore::IR::RegisterClassType, IROps::OP_LAST + 1> IRRegClasses = {\n")
for op in IROps:
if op.Name == "Last":
output_file.write("\tRegClass::Invalid,\n")
output_file.write("\tFEXCore::IR::InvalidClass,\n")
else:
if op.HasDest and op.DestType is None:
Class = "Invalid"
if op.HasDest and op.DestType == None:
ExitError("IR op {} has destination with no destination class".format(op.Name))
if op.HasDest and op.DestType == "SSA": # Special case SSA type
output_file.write("\tRegClass::Complex,\n")
output_file.write("\tFEXCore::IR::ComplexClass,\n")
elif op.HasDest:
output_file.write("\tRegClass::{},\n".format(op.DestType))
output_file.write("\tFEXCore::IR::{}Class,\n".format(op.DestType))
else:
# No destination so it has an invalid destination class
output_file.write("\tRegClass::Invalid, // No destination\n")
output_file.write("\tFEXCore::IR::InvalidClass, // No destination\n")
output_file.write("};\n\n")
output_file.write("// Make sure our array maps directly to the IROps enum\n")
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == RegClass::Invalid);\n\n")
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == FEXCore::IR::InvalidClass);\n\n")
output_file.write("FEXCore::IR::RegClass GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
output_file.write("FEXCore::IR::RegisterClassType GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
output_file.write("#undef IROP_REG_CLASSES_IMPL\n")
output_file.write("#endif\n\n")
@@ -566,7 +568,9 @@ def print_ir_arg_printer():
SSAArgNum = 0
FirstArg = True
for arg in op.Arguments:
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
# No point printing temporaries that we can't recover
if arg.Temporary:
continue
@@ -667,7 +671,7 @@ def print_ir_allocator_helpers():
output_file.write("\t\treturn HeaderOp->Op;\n")
output_file.write("\t}\n\n")
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n")
output_file.write("\tFEXCore::IR::RegisterClassType GetOpRegClass(const OrderedNode *Op) const {\n")
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
output_file.write("\t}\n\n")
@@ -681,21 +685,22 @@ def print_ir_allocator_helpers():
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
# Output SSA args first
for i, arg in enumerate(op.Arguments):
LastArg = i == len(op.Arguments) - 1
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
if arg.Temporary:
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
elif arg.IsSSA:
# SSA value
output_file.write("OrderedNodeWrapper {}".format(arg.Name))
else:
# User defined op that is stored
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
if arg.DefaultInitializer:
if arg.DefaultInitializer != None:
output_file.write(" = {}".format(arg.DefaultInitializer))
if not LastArg:
@@ -753,19 +758,20 @@ def print_ir_allocator_helpers():
if op.SSAArgNum:
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
for i, arg in enumerate(op.Arguments):
LastArg = i == len(op.Arguments) - 1
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
if arg.Temporary:
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
elif arg.IsSSA:
output_file.write("OrderedNode *{}".format(arg.Name))
else:
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
if arg.DefaultInitializer:
if arg.DefaultInitializer != None:
output_file.write(" = {}".format(arg.DefaultInitializer))
if not LastArg:
@@ -806,15 +812,16 @@ def print_ir_allocator_helpers():
print_validation(op)
output_file.write(f"\t\treturn _{op.Name}(")
for i, arg in enumerate(op.Arguments):
LastArg = i == len(op.Arguments) - 1
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
output_file.write(arg.Name)
if arg.IsSSA:
output_file.write("->Wrapped(ListDataBegin)")
if not LastArg:
output_file.write(", ")
output_file.write(");\n")
output_file.write("\t}\n\n")
output_file.write(");\n");
output_file.write("\t}\n\n");
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
output_file.write("#endif\n")
@@ -845,8 +852,8 @@ def print_ir_dispatcher_dispatch():
output_dispatch_file.write("#endif\n")
if len(sys.argv) < 4:
ExitError("Insufficient parameters passed to script")
if (len(sys.argv) < 4):
ExitError()
output_filename = sys.argv[2]
output_dispatcher_filename = sys.argv[3]
@@ -858,7 +865,6 @@ json_file.close()
json_object = json.loads(json_text)
json_object = {k.upper(): v for k, v in json_object.items()}
enums = json_object["ENUMS"]
ops = json_object["OPS"]
irtypes = json_object["IRTYPES"]
defines = json_object["DEFINES"]
@@ -868,7 +874,7 @@ parse_ops(ops)
output_file = open(output_filename, "w")
print_enums(enums)
print_enums()
print_ir_structs(defines)
print_ir_sizes()
print_ir_reg_classes()
+2 -2
View File
@@ -18,7 +18,6 @@ set (SRCS
Common/JitSymbols.cpp
Interface/Context/Context.cpp
Interface/Core/LookupCache.cpp
Interface/Core/CodeCache.cpp
Interface/Core/Core.cpp
Interface/Core/CPUBackend.cpp
Interface/Core/Addressing.cpp
@@ -58,6 +57,7 @@ set (SRCS
Interface/Core/X86Tables/VEXTables.cpp
Interface/Core/X86Tables/X87Tables.cpp
Interface/GDBJIT/GDBJIT.cpp
Interface/IR/AOTIR.cpp
Interface/IR/IRDumper.cpp
Interface/IR/IREmitter.cpp
Interface/IR/PassManager.cpp
@@ -202,7 +202,7 @@ add_custom_target(CONFIG_INC
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
# Add in diagnostic colours if the option is available.
# Ninja code generator will kill colours if this isn't here
+8 -82
View File
@@ -4,9 +4,9 @@
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/BitUtils.h>
#include "cephes_128bit.h"
#include <bit>
#include <cmath>
#include <cstring>
#include <stdint.h>
@@ -501,53 +501,12 @@ struct FEX_PACKED X80SoftFloat {
float ToF32(softfloat_state* state) const {
const float32_t Result = extF80_to_f32(state, *this);
return std::bit_cast<float>(Result);
}
bool IsSignalingNaN() const {
return (Exponent == 0x7FFF) && (Significand & 0x8000000000000000ULL) && !(Significand & 0x4000000000000000ULL) && // Bit 62 clear (signaling)
(Significand & 0x3FFFFFFFFFFFFFFFULL);
}
bool IsQuietNaN() const {
return (Exponent == 0x7FFF) && (Significand & 0x8000000000000000ULL) && (Significand & 0x4000000000000000ULL); // Bit 62 set (quiet)
}
// Helper to detect if this is any NaN
bool IsNaN() const {
return IsSignalingNaN() || IsQuietNaN();
}
// X87 value to F64 while preserving signaling nan property
double ToF64_PreserveNan(softfloat_state* state) const {
if (IsSignalingNaN()) {
// we keep it as a signaling nan in ieee754 in 64bits
uint64_t sign_bit = Sign ? 0x8000000000000000ULL : 0;
uint64_t exp_bits = 0x7FF0000000000000ULL;
uint64_t x87_frac = Significand & 0x3FFFFFFFFFFFFFFFULL;
uint64_t ieee_frac = (x87_frac >> 11) & 0x0007FFFFFFFFFFFFULL;
if (ieee_frac == 0) {
ieee_frac = 1;
}
ieee_frac &= ~0x0008000000000000ULL;
uint64_t result_bits = sign_bit | exp_bits | ieee_frac;
return std::bit_cast<double>(result_bits);
} else if (IsQuietNaN()) {
const float64_t Result = extF80_to_f64(state, *this);
uint64_t result_bits = std::bit_cast<uint64_t>(Result);
result_bits |= 0x0008000000000000ULL;
return std::bit_cast<double>(result_bits);
} else {
const float64_t Result = extF80_to_f64(state, *this);
return std::bit_cast<double>(Result);
}
return FEXCore::BitCast<float>(Result);
}
double ToF64(softfloat_state* state) const {
const float64_t Result = extF80_to_f64(state, *this);
return std::bit_cast<double>(Result);
return FEXCore::BitCast<double>(Result);
}
FEXCore::VectorRegType ToVector() const {
@@ -559,7 +518,7 @@ struct FEX_PACKED X80SoftFloat {
BIGFLOAT ToFMax(softfloat_state* state) const {
#if BIGFLOATSIZE == 16
const float128_t Result = extF80_to_f128(state, *this);
return std::bit_cast<BIGFLOAT>(Result);
return FEXCore::BitCast<BIGFLOAT>(Result);
#else
BIGFLOAT result {};
memcpy(&result, this, sizeof(result));
@@ -618,51 +577,18 @@ struct FEX_PACKED X80SoftFloat {
}
X80SoftFloat(softfloat_state* state, const float rhs) {
*this = f32_to_extF80(state, std::bit_cast<float32_t>(rhs));
*this = f32_to_extF80(state, FEXCore::BitCast<float32_t>(rhs));
}
X80SoftFloat(softfloat_state* state, const double rhs) {
*this = f64_to_extF80(state, std::bit_cast<float64_t>(rhs));
}
// Create X80SoftFloat from double while preserving NaN signaling properties
static X80SoftFloat FromF64_PreserveNaN(softfloat_state* state, double value) {
uint64_t bits = std::bit_cast<uint64_t>(value);
// Check if it's a nan
if ((bits & 0x7FF0000000000000ULL) == 0x7FF0000000000000ULL && (bits & 0x000FFFFFFFFFFFFFULL) != 0) {
X80SoftFloat result;
result.Sign = (bits >> 63) & 1;
result.Exponent = 0x7FFF;
bool is_signaling = !(bits & 0x0008000000000000ULL);
uint64_t ieee_payload = bits & 0x0007FFFFFFFFFFFFULL;
// set bit 63 required for x87
result.Significand = 0x8000000000000000ULL;
if (is_signaling) { // clear bit 62 for signaling nan
result.Significand &= ~0x4000000000000000ULL;
} else { // clear bit 62 for quiet nan
result.Significand |= 0x4000000000000000ULL;
}
// ieee754 51-bit payload -> x87 62-bit payload
result.Significand |= (ieee_payload << 11) & 0x3FFFFFFFFFFFFFFFULL;
return result;
}
// For non-NaN values, use standard conversion
return X80SoftFloat(state, value);
*this = f64_to_extF80(state, FEXCore::BitCast<float64_t>(rhs));
}
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
#if BIGFLOATSIZE == 16
*this = f128_to_extF80(state, std::bit_cast<float128_t>(rhs));
*this = f128_to_extF80(state, FEXCore::BitCast<float128_t>(rhs));
#else
*this = std::bit_cast<long double>(rhs);
*this = FEXCore::BitCast<long double>(rhs);
#endif
}
+47 -11
View File
@@ -2,23 +2,59 @@
#pragma once
#include <FEXCore/fextl/string.h>
#include <concepts>
#include <cstdint>
#include <string_view>
#include <optional>
namespace FEXCore::StrConv {
template<std::integral T>
bool Conv(std::string_view Value, T* Result) {
if constexpr (std::is_signed_v<T>) {
*Result = static_cast<T>(std::strtoll(Value.data(), nullptr, 0));
} else {
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
}
inline bool Conv(std::string_view Value, bool* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
template<typename T, typename = std::enable_if_t<std::is_enum_v<T>, T>>
bool Conv(std::string_view Value, T* Result) {
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
inline bool Conv(std::string_view Value, uint8_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int8_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint16_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int16_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint32_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int32_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint64_t* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int64_t* Result) {
*Result = std::strtoll(Value.data(), nullptr, 0);
return true;
}
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
inline bool Conv(std::string_view Value, T* Result) {
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
return true;
}
+28 -13
View File
@@ -4,6 +4,7 @@
"Multiblock": {
"Type": "bool",
"Default": "true",
"ShortArg": "m",
"Desc": [
"Controls multiblock code compilation",
"Can cause long JIT compilation times and stutter"
@@ -12,6 +13,7 @@
"MaxInst": {
"Type": "int32",
"Default": "5000",
"ShortArg": "n",
"Desc": [
"Maximum number of instruction to store in a block"
]
@@ -59,9 +61,7 @@
"ENABLEWFXT": "enablewfxt",
"DISABLEWFXT": "disablewfxt",
"ENABLE3DNOW": "enable3dnow",
"DISABLE3DNOW": "disable3dnow",
"ENABLESSE4A": "enablesse4a",
"DISABLESSE4A": "disablesse4a"
"DISABLE3DNOW": "disable3dnow"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
@@ -84,8 +84,7 @@
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it"
"\t{enable,disable}3dnow: Will force enable or disable 3DNow even if the host doesn't support it"
]
},
"SmallTSCScale": {
@@ -100,6 +99,7 @@
"RootFS": {
"Type": "str",
"Default": "",
"ShortArg": "R",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
@@ -114,6 +114,7 @@
"ThunkHostLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
]
@@ -121,6 +122,7 @@
"ThunkGuestLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
@@ -128,6 +130,7 @@
"ThunkConfig": {
"Type": "str",
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks.",
"This can be a filesystem path",
@@ -142,6 +145,7 @@
"Env": {
"Type": "strarray",
"Default": "",
"ShortArg": "E",
"Desc": [
"Adds an environment variable to the emulated environment."
]
@@ -149,6 +153,7 @@
"HostEnv": {
"Type": "strarray",
"Default": "",
"ShortArg": "H",
"Desc": [
"Adds an environment variable to the host environment.",
"This can be useful for setting environment variables that thunks can pick up.",
@@ -167,6 +172,7 @@
"SingleStep": {
"Type": "bool",
"Default": "false",
"ShortArg": "S",
"Desc": [
"Single stepping configuration."
]
@@ -174,6 +180,7 @@
"GdbServer": {
"Type": "bool",
"Default": "false",
"ShortArg": "G",
"Desc": [
"Enables the GDB server."
]
@@ -207,6 +214,7 @@
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"ShortArg": "g",
"Desc": [
"When the test harness ends, print the GPR state."
]
@@ -214,6 +222,7 @@
"O0": {
"Type": "bool",
"Default": "false",
"ShortArg": "O0",
"Desc": [
"Disables optimizations passes for debugging."
]
@@ -303,6 +312,7 @@
"SilentLog": {
"Type": "bool",
"Default": "true",
"ShortArg": "s",
"Desc": [
"Disables logging"
]
@@ -310,6 +320,7 @@
"OutputLog": {
"Type": "str",
"Default": "server",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
"[stdout, stderr, server, <Filename>]"
@@ -384,6 +395,14 @@
"This is required to ensure a split-lock doesn't tear inside the process"
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
"Desc": [
"Automatically enables TSO when shared memory is used.",
"Should work without issues in most cases."
]
},
"VolatileMetadata": {
"Type": "bool",
"Default": "true",
@@ -399,14 +418,6 @@
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
]
},
"X87StrictReducedPrecision": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enables stricter X87 floating point behavior when X87ReducedPrecision is enabled.",
"Adds additional checks and implementations like NaN propagation for better compatibility."
]
},
"ABILocalFlags": {
"Type": "bool",
"Default": "false",
@@ -503,6 +514,10 @@
},
"UnnamedOptions": {
"Misc": {
"IS_INTERPRETER": {
"Type": "bool",
"Default": "false"
},
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false"
@@ -1,7 +1,6 @@
// SPDX-License-Identifier: MIT
#include "Interface/Context/Context.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Core/CoreState.h>
+44 -45
View File
@@ -5,46 +5,53 @@
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/AOTIR.h"
#include <Interface/IR/IntrusiveIRList.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <stdint.h>
#include <atomic>
#include <cstddef>
#include <cstdint>
#include <mutex>
#include <optional>
#include <shared_mutex>
namespace FEXCore {
class SignalDelegator;
class CodeLoader;
class ThunkHandler;
namespace Core {
struct DebugData;
struct InternalThreadState;
} // namespace Core
namespace CPU {
class Arm64JITCore;
class Dispatcher;
} // namespace CPU
namespace HLE {
class SourcecodeResolver;
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
} // namespace HLE
} // namespace FEXCore
namespace FEXCore::IR {
namespace Validation {
class IRValidation;
}
} // namespace FEXCore::IR
namespace FEXCore::Context {
struct FEX_PACKED ExitFunctionLinkData {
uint64_t HostCode;
@@ -64,23 +71,7 @@ struct CustomIRResult {
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
class CodeCache : public AbstractCodeCache {
public:
CodeCache(ContextImpl&);
~CodeCache();
ContextImpl& CTX;
bool IsGeneratingCache = false;
void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
void InitiateCacheGeneration() override {
IsGeneratingCache = true;
}
};
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
public:
// Context base class implementation.
bool InitCore() override;
@@ -150,9 +141,10 @@ public:
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
CodeCache& GetCodeCache() override {
return CodeCache;
}
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
void FinalizeAOTIRCache() override {}
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
@@ -162,6 +154,8 @@ public:
return CodeInvalidationMutex;
}
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
@@ -183,6 +177,13 @@ public:
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
public:
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
friend class FEXCore::IR::Validation::IRValidation;
struct {
uint64_t VirtualMemSize {1ULL << 36};
uint64_t TSCScale = 0;
@@ -195,6 +196,7 @@ public:
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
@@ -207,7 +209,6 @@ public:
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(x87StrictReducedPrecision, X87STRICTREDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
@@ -226,7 +227,6 @@ public:
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
FEXCore::ThunkHandler* ThunkHandler {};
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CodeCache CodeCache;
SignalDelegator* SignalDelegation {};
X86GeneratedCode X86CodeGen;
@@ -317,16 +317,12 @@ protected:
VectorAtomicTSOEmulationEnabled = true;
MemcpyAtomicTSOEmulationEnabled = true;
} else {
AtomicTSOEmulationEnabled = Config.TSOEnabled;
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
MemcpyAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.MemcpySetTSOEnabled;
}
}
void UpdateX87PrecisionConfig() {
// If strict reduced precision is enabled, automatically enable reduced precision
if (Config.x87StrictReducedPrecision() && !Config.x87ReducedPrecision()) {
FEXCore::Config::Set(FEXCore::Config::CONFIG_X87REDUCEDPRECISION, "1");
// Atomic TSO emulation only enabled if the config option is enabled.
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
}
}
@@ -340,6 +336,9 @@ private:
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
IR::AOTIRCaptureCache IRCaptureCache;
bool IsMemoryShared = false;
bool SupportsHardwareTSO = false;
bool AtomicTSOEmulationEnabled = true;
bool VectorAtomicTSOEmulationEnabled = false;
@@ -352,8 +351,8 @@ private:
std::atomic<bool> HasCustomIRHandlers {};
struct CustomIRHandlerEntry final {
CustomIREntrypointHandler Handler;
void* Creator;
void* Data;
void *Creator;
void *Data;
};
fextl::unordered_map<uint64_t, CustomIRHandlerEntry> CustomIRHandlers;
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
+11 -13
View File
@@ -7,7 +7,7 @@
namespace FEXCore::IR {
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
Ref Tmp = A.Base;
if (A.Offset) {
@@ -51,8 +51,8 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
return Tmp ?: IREmit->Constant(0);
}
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
bool Vector, IR::OpSize AccessSize) {
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
IR::OpSize AccessSize) {
const auto Is32Bit = GPRSize == OpSize::i32Bit;
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
@@ -103,7 +103,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset),
.IndexType = MemOffsetType::SXTX,
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = 1,
};
}
@@ -111,17 +111,15 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
if (AtomicTSO) {
// TODO: LRCPC3 support for vector Imm9.
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
AddressMode B = A;
// ScaledRegisterLoadstore
if (B.Index && B.Segment) {
B.Base = IREmit->Add(GPRSize, B.Base, B.Segment);
} else if (B.Segment) {
B.Index = B.Segment;
B.IndexScale = 1;
if (A.Index && A.Segment) {
A.Base = IREmit->Add(GPRSize, A.Base, A.Segment);
} else if (A.Segment) {
A.Index = A.Segment;
A.IndexScale = 1;
}
return B;
return A;
}
if (Vector || !AtomicTSO) {
@@ -136,7 +134,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset),
.IndexType = MemOffsetType::SXTX,
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = 1,
};
}
+6 -7
View File
@@ -11,18 +11,17 @@ struct AddressMode {
Ref Segment {nullptr};
Ref Base {nullptr};
Ref Index {nullptr};
int64_t Offset = 0;
MemOffsetType IndexType = MemOffsetType::SXTX;
MemOffsetType IndexType = MEM_OFFSET_SXTX;
uint8_t IndexScale = 1;
int64_t Offset = 0;
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
IR::OpSize AddrSize;
bool NonTSO;
};
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
bool Vector, IR::OpSize AccessSize);
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
IR::OpSize AccessSize);
} // namespace FEXCore::IR
}; // namespace FEXCore::IR
@@ -1,10 +1,10 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "FEXCore/Core/X86Enums.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Context/Context.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
@@ -1,31 +1,30 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "FEXCore/Utils/EnumUtils.h"
#include "Interface/Core/JIT/Relocations.h"
#ifdef VIXL_DISASSEMBLER
#include <aarch64/disasm-aarch64.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/vector.h>
#endif
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#include <aarch64/simulator-constants-aarch64.h>
#endif
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/vector.h>
#include <CodeEmitter/Emitter.h>
#include <CodeEmitter/Registers.h>
#include <cstddef>
#include <cstdint>
#include <optional>
#include <span>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::X86State {
enum X86Reg : uint32_t;
}
namespace FEXCore::CPU {
// Contains the address to the currently available CPU state
+4 -11
View File
@@ -1,17 +1,14 @@
// SPDX-License-Identifier: MIT
#include "FEXCore/IR/IR.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/Utils/PrctlUtils.h>
#include <cstdint>
#include "LookupCache.h"
#ifndef _WIN32
#include <linux/prctl.h>
#include <sys/prctl.h>
#endif
@@ -360,10 +357,6 @@ namespace CPU {
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
}
#ifndef _WIN32
prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, "FEXMemJIT");
#endif
LookupCache = fextl::make_unique<GuestToHostMap>();
}
+12 -5
View File
@@ -17,10 +17,6 @@ $end_info$
#include <cstdint>
namespace FEXCore::CPU {
union Relocation;
}
namespace FEXCore {
namespace IR {
@@ -161,7 +157,18 @@ namespace CPU {
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() = 0;
/**
* @brief Relocates a block of code from the JIT code object cache
*
* @param Entry - RIP of the entry
* @param SerializationData - Serialization data referring to the object cache for `Entry`
*
* @return An executable function pointer relocated from the cache object
*/
[[nodiscard]]
virtual void* RelocateJITObjectCode(uint64_t /* Entry */, const CodeSerialize::CodeObjectFileSection* /* SerializationData */) {
return nullptr;
}
virtual void ClearCache() {}
+64 -81
View File
@@ -43,15 +43,12 @@ namespace ProductNames {
static const char ARM_A715[] = "Cortex-A715";
static const char ARM_A720[] = "Cortex-A720";
static const char ARM_A725[] = "Cortex-A725";
static const char ARM_C1Pro[] = "C1-Pro";
static const char ARM_C1Premium[] = "C1-Premium";
static const char ARM_X1[] = "Cortex-X1";
static const char ARM_X1C[] = "Cortex-X1C";
static const char ARM_X2[] = "Cortex-X2";
static const char ARM_X3[] = "Cortex-X3";
static const char ARM_X4[] = "Cortex-X4";
static const char ARM_X925[] = "Cortex-X925";
static const char ARM_C1Ultra[] = "C1-Ultra";
static const char ARM_N1[] = "Neoverse N1";
static const char ARM_N2[] = "Neoverse N2";
static const char ARM_N3[] = "Neoverse N3";
@@ -62,7 +59,6 @@ namespace ProductNames {
static const char ARM_A65[] = "Cortex-A65";
static const char ARM_A510[] = "Cortex-A510";
static const char ARM_A520[] = "Cortex-A520";
static const char ARM_C1Nano[] = "C1-Nano";
static const char ARM_Kryo200[] = "Kryo 2xx";
static const char ARM_Kryo300[] = "Kryo 3xx";
@@ -74,7 +70,6 @@ namespace ProductNames {
static const char ARM_Denver[] = "Nvidia Denver";
static const char ARM_Carmel[] = "Nvidia Carmel";
static const char ARM_Olympus[] = "Nvidia Olympus";
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
@@ -90,9 +85,6 @@ namespace ProductNames {
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
static const char ARM_ORYON_1[] = "Oryon-1";
static const char ARM_Ampere_1[] = "AmpereOne";
static const char ARM_Ampere_1A[] = "AmpereOneA";
static const char ARM_Ampere_1B[] = "AmpereOneB";
#else
#endif
} // namespace ProductNames
@@ -178,7 +170,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
// CPU priority order
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
// Relative list so things they will commonly end up in big.little configurations sort of relate
static constexpr std::array<CPUMIDR, 66> CPUMIDRs = {{
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
// Typically big CPU cores
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
@@ -189,46 +181,38 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
{0x41, 0xd8b, 1, ProductNames::ARM_C1Pro}, // C1-Pro
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0xc0, 0xac3, 1, ProductNames::ARM_Ampere_1}, // AmpereOne
{0xc0, 0xac4, 1, ProductNames::ARM_Ampere_1A}, // AmpereOneA
{0xc0, 0xac5, 1, ProductNames::ARM_Ampere_1B}, // AmpereOneB
{0x4e, 0x010, 1, ProductNames::ARM_Olympus}, // Olympus
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
// Denver rated above A57 to match TX2 weirdness
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
@@ -243,7 +227,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
{0x41, 0xd8a, 1, ProductNames::ARM_C1Nano}, // C1-Nano
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
@@ -909,38 +892,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
Res.eax = FAMILY_IDENTIFIER;
Res.ecx = (1 << 0) | // LAHF/SAHF
(1 << 1) | // 0 = Single core product, 1 = multi core product
(0 << 2) | // SVM
(1 << 3) | // Extended APIC register space
(0 << 4) | // LOCK MOV CR0 means MOV CR8
(1 << 5) | // ABM instructions
(CTX->HostFeatures.SupportsSSE4a << 6) | // SSE4a
(0 << 7) | // Misaligned SSE mode
(1 << 8) | // PREFETCHW
(0 << 9) | // OS visible workaround support
(0 << 10) | // Instruction based sampling support
(0 << 11) | // XOP
(0 << 12) | // SKINIT
(0 << 13) | // Watchdog timer support
(0 << 14) | // Reserved
(0 << 15) | // Lightweight profiling support
(0 << 16) | // FMA4
(1 << 17) | // Translation cache extension
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(0 << 20) | // Reserved
(0 << 21) | // XOP-TBM
(0 << 22) | // Topology extensions support
(0 << 23) | // Core performance counter extensions
(0 << 24) | // NB performance counter extensions
(0 << 25) | // Reserved
(0 << 26) | // Data breakpoints extensions
(0 << 27) | // Performance TSC
(0 << 28) | // L2 perf counter extensions
(0 << 29) | // MONITORX
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.ecx = (1 << 0) | // LAHF/SAHF
(1 << 1) | // 0 = Single core product, 1 = multi core product
(0 << 2) | // SVM
(1 << 3) | // Extended APIC register space
(0 << 4) | // LOCK MOV CR0 means MOV CR8
(1 << 5) | // ABM instructions
(0 << 6) | // SSE4a
(0 << 7) | // Misaligned SSE mode
(1 << 8) | // PREFETCHW
(0 << 9) | // OS visible workaround support
(0 << 10) | // Instruction based sampling support
(0 << 11) | // XOP
(0 << 12) | // SKINIT
(0 << 13) | // Watchdog timer support
(0 << 14) | // Reserved
(0 << 15) | // Lightweight profiling support
(0 << 16) | // FMA4
(1 << 17) | // Translation cache extension
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(0 << 20) | // Reserved
(0 << 21) | // XOP-TBM
(0 << 22) | // Topology extensions support
(0 << 23) | // Core performance counter extensions
(0 << 24) | // NB performance counter extensions
(0 << 25) | // Reserved
(0 << 26) | // Data breakpoints extensions
(0 << 27) | // Performance TSC
(0 << 28) | // L2 perf counter extensions
(0 << 29) | // MONITORX
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.edx = (1 << 0) | // FPU
(1 << 1) | // Virtual mode extensions
@@ -1,27 +0,0 @@
// SPDX-License-Identifier: MIT
#include <Interface/Context/Context.h>
#include <FEXCore/HLE/SourcecodeResolver.h>
namespace FEXCore {
ExecutableFileInfo::~ExecutableFileInfo() = default;
} // namespace FEXCore
namespace FEXCore::Context {
CodeCache::CodeCache(ContextImpl& CTX_)
: CTX(CTX_) {}
CodeCache::~CodeCache() = default;
void CodeCache::LoadData(Core::InternalThreadState& Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& GuestRIPLookup) {
// TODO
}
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
// TODO
return true;
}
} // namespace FEXCore::Context
+63 -43
View File
@@ -18,7 +18,6 @@ $end_info$
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <Interface/GDBJIT/GDBJIT.h>
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
@@ -57,13 +56,20 @@ $end_info$
#include <algorithm>
#include <array>
#include <atomic>
#include <chrono>
#include <condition_variable>
#include <fcntl.h>
#include <functional>
#include <mutex>
#include <queue>
#include <shared_mutex>
#include <signal.h>
#include <stdio.h>
#include <string_view>
#include <sys/stat.h>
#include <type_traits>
#include <unistd.h>
#include <unordered_map>
#include <utility>
#include <xxhash.h>
@@ -71,7 +77,7 @@ namespace FEXCore::Context {
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
: HostFeatures {Features}
, CPUID {this}
, CodeCache {*this} {
, IRCaptureCache {this} {
if (!Config.Is64BitMode()) {
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
Config.VirtualMemSize = 1ULL << 32;
@@ -93,8 +99,6 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
// Track atomic TSO emulation configuration.
UpdateAtomicTSOEmulationConfig();
// Ensure X87 precision constraints are respected.
UpdateX87PrecisionConfig();
}
struct GetFrameBlockInfoResult {
@@ -606,7 +610,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
} else {
ForceTSO = IR::ForceTSOMode::ForceDisabled;
}
} else if (DecodedInfo->Flags & X86Tables::DecodeFlags::FLAG_FORCE_TSO) {
} else if (DecodedInfo->ForceTSO) {
ForceTSO = IR::ForceTSOMode::ForceEnabled;
}
@@ -704,9 +708,9 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
if (SourcecodeResolver && Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
if (MappedSection) {
MappedSection->FileInfo.SourcecodeMap = SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, MappedSection->FileInfo.FileId);
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (AOTIRCacheEntry.Entry) {
AOTIRCacheEntry.Entry->SourcecodeMap = SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
}
}
@@ -781,44 +785,36 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
if (Config.BlockJITNaming()) {
auto FragmentBasePtr = CompiledCode.BlockBegin;
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
if (DebugData) {
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
}
}
}
} else {
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
}
}
}
if (Config.LibraryJITNaming() || Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
if (MappedSection) {
if (Config.LibraryJITNaming()) {
Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, MappedSection->FileInfo.Filename);
}
if (Config.GDBSymbols()) {
GDBJITRegister(MappedSection->FileInfo, MappedSection->FileStartVA, GuestRIP, (uintptr_t)CodePtr, *DebugData);
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
}
}
}
}
// Clear any relocations that might have been generated
if (!CodeCache.IsGeneratingCache) {
Thread->CPUBackend->ClearRelocations();
Thread->CPUBackend->ClearRelocations();
if (IRCaptureCache.PostCompileCode(Thread, CompiledCode.BlockBegin, GuestRIP, StartAddr, Length, DebugData.get())) {
// Early exit
return (uintptr_t)CodePtr;
}
if (NeedsAddGuestCodeRanges) {
@@ -897,6 +893,23 @@ void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* T
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
}
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
if (!Thread) {
return;
}
if (!IsMemoryShared) {
IsMemoryShared = true;
UpdateAtomicTSOEmulationConfig();
if (Config.TSOAutoMigration) {
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
Thread->LookupCache->ClearCache();
}
}
}
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
"be unique_locked here");
@@ -947,10 +960,10 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
if (GPRSize == IR::OpSize::i64Bit) {
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
R->Reg = IR::PhysicalRegister(IR::RegClass::GPRFixed, X86State::REG_R11).Raw;
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
} else {
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
offsetof(Core::CPUState, mm[0][0]));
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
offsetof(Core::CPUState, mm[0][0]));
}
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
},
@@ -1002,9 +1015,9 @@ void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint
auto lk = GuardSignalDeferringSection(CTX->CodeInvalidationMutex, Thread);
if (Size == 8) {
*reinterpret_cast<uint64_t*>(Address) = Value;
*reinterpret_cast<uint64_t *>(Address) = Value;
} else if (Size == 4) {
*reinterpret_cast<uint32_t*>(Address) = Value;
*reinterpret_cast<uint32_t *>(Address) = Value;
} else {
ERROR_AND_DIE_FMT("Unexpected write size for backpatcher: {}", Size);
}
@@ -1013,6 +1026,13 @@ void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint
CTX->SyscallHandler->InvalidateGuestCodeRange(Thread, Address, Size);
}
IR::AOTIRCacheEntry* ContextImpl::LoadAOTIRCacheEntry(const fextl::string& filename) {
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
return rv;
}
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {}
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
@@ -17,7 +17,6 @@
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <CodeEmitter/Emitter.h>
+31 -39
View File
@@ -259,13 +259,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
if (HasSIB) {
FEXCore::X86Tables::SIBDecoded SIB;
if (DecodeInst->Flags & DecodeFlags::FLAG_DECODED_SIB) {
if (DecodeInst->DecodedSIB) {
SIB.Hex = DecodeInst->SIB;
} else {
// Haven't yet grabbed SIB, pull it now
DecodeInst->SIB = ReadByte();
SIB.Hex = DecodeInst->SIB;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_SIB;
DecodeInst->DecodedSIB = true;
}
// If the SIB base is 0b101, aka BP or R13 then we have a 32bit displacement
@@ -401,9 +401,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// If we require ModRM and haven't decoded it yet, do it now
// Some instructions have to read modrm upfront, others do it later
if (HasMODRM && !(DecodeInst->Flags & DecodeFlags::FLAG_DECODED_MODRM)) {
if (HasMODRM && !DecodeInst->DecodedModRM) {
DecodeInst->ModRM = ReadByte();
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
}
// New instruction size decoding
@@ -436,8 +436,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_16BIT);
DestSize = 2;
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) && (HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) &&
(HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_64BIT);
DestSize = 8;
} else {
@@ -464,8 +465,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// See table 1-2. Operand-Size Overrides for this decoding
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) && (HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) &&
(HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_64BIT);
} else {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_32BIT);
@@ -634,20 +636,11 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
Literal = static_cast<int32_t>(Literal);
}
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
DecodeInst->Src[CurrentSrc].Data.Literal.SignExtend = true;
}
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
++CurrentSrc;
if (Bytes == 8) [[unlikely]] {
DecodeInst->Src[CurrentSrc].Data.Literal.Size = 4;
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal >> 32;
}
Bytes = 0;
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
}
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
@@ -679,7 +672,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
} else if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -696,18 +689,18 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
constexpr uint16_t PF_F2 = 3;
uint16_t PrefixType = PF_NONE;
if (LastEscapePrefix == 0xF3) {
if (DecodeInst->LastEscapePrefix == 0xF3) {
PrefixType = PF_F3;
} else if (LastEscapePrefix == 0xF2) {
} else if (DecodeInst->LastEscapePrefix == 0xF2) {
PrefixType = PF_F2;
} else if (LastEscapePrefix == 0x66) {
} else if (DecodeInst->LastEscapePrefix == 0x66) {
PrefixType = PF_66;
}
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -734,7 +727,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
return NormalOp(&(*X87Table)[X87Op], X87Op);
@@ -796,7 +789,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -820,7 +813,6 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
InstructionSize = 0;
LastEscapePrefix = 0;
Instruction.fill(0);
DecodeInst = &DecodedBuffer[DecodedSize];
@@ -842,7 +834,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
// Decode ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -881,7 +873,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
uint16_t LocalOp = (Prefix << 8) | ReadByte();
bool NoOverlay66 = (FEXCore::X86Tables::H0F38TableOps[LocalOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
if (LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
if (DecodeInst->LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather than modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
@@ -897,7 +889,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
constexpr uint16_t PF_3A_REX = (1 << 1);
uint16_t Prefix = PF_3A_NONE;
if (LastEscapePrefix == 0x66) { // Operand Size
if (DecodeInst->LastEscapePrefix == 0x66) { // Operand Size
Prefix = PF_3A_66;
}
@@ -923,17 +915,17 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
if (NoOverlay) { // This section of the table ignores prefix extention
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0xF3) { // REP
} else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0xF2) { // REPNE
} else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
} else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
@@ -949,7 +941,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
}
case 0x66: // Operand Size prefix
DecodeInst->Flags |= DecodeFlags::FLAG_OPERAND_SIZE;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
break;
case 0x67: // Address Size override prefix
@@ -980,11 +972,11 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
break;
case 0xF2: // REPNE prefix
DecodeInst->Flags |= DecodeFlags::FLAG_REPNE_PREFIX;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
break;
case 0xF3: // REP prefix
DecodeInst->Flags |= DecodeFlags::FLAG_REP_PREFIX;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
break;
case 0x64: // FS prefix
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_FS_PREFIX;
@@ -1063,10 +1055,10 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
if (DecodeInst->OP == 0x8b && DecodeInst->Src[0].IsGPRIndirect() &&
IsKnownAtomicDisplacement(DecodeInst->Src[0].Data.GPRIndirect.Displacement)) {
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
DecodeInst->ForceTSO = true;
}
if (DecodeInst->OP == 0x89 && DecodeInst->Dest.IsGPRIndirect() && IsKnownAtomicDisplacement(DecodeInst->Dest.Data.GPRIndirect.Displacement)) {
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
DecodeInst->ForceTSO = true;
}
}
@@ -1309,7 +1301,7 @@ const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, u
return _InstStream - EntryPoint + RIP;
}
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
BlockInfo.TotalInstructionCount = 0;
BlockInfo.Blocks.clear();
+1 -2
View File
@@ -49,7 +49,7 @@ public:
};
Decoder(FEXCore::Core::InternalThreadState* Thread);
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
const DecodedBlockInformation* GetDecodedBlockInfo() const {
return &BlockInfo;
@@ -125,7 +125,6 @@ private:
static constexpr size_t MAX_INST_SIZE = 15;
uint8_t InstructionSize {};
std::array<uint8_t, MAX_INST_SIZE> Instruction;
uint8_t LastEscapePrefix {};
FEXCore::X86Tables::DecodedInst* DecodeInst;
// This is for multiblock data tracking
@@ -2,13 +2,11 @@
#pragma once
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
#include "Interface/IR/IR.h"
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/SHMStats.h>
#include <FEXCore/Config/Config.h>
namespace FEXCore::CPU {
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
@@ -79,12 +77,6 @@ struct OpHandlers<IR::OP_F80CVTTO> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle8(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
auto Context = static_cast<Context::ContextImpl*>(Frame->Thread->CTX);
auto ReducedPrecisionMode = Context->Config.x87ReducedPrecision;
auto StrictReducedPrecisionMode = Context->Config.x87StrictReducedPrecision;
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
return X80SoftFloat::FromF64_PreserveNaN(&State.State, src);
}
return X80SoftFloat(&State.State, src);
}
};
@@ -123,12 +115,6 @@ struct OpHandlers<IR::OP_F80CVT> {
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
auto Context = static_cast<Context::ContextImpl*>(Frame->Thread->CTX);
auto ReducedPrecisionMode = Context->Config.x87ReducedPrecision;
auto StrictReducedPrecisionMode = Context->Config.x87StrictReducedPrecision;
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
return X80SoftFloat(src).ToF64_PreserveNan(&State.State);
}
return X80SoftFloat(src).ToF64(&State.State);
}
};
@@ -87,10 +87,12 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SINCOS>::handle)};
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
// Double Precision Binary
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle)};
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
@@ -218,21 +220,21 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
return true; \
}
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
return true; \
}
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64x2_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
return true; \
}
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
return true; \
}
// Unary
+18 -36
View File
@@ -372,7 +372,7 @@ DEF_OP(CondSubNZCV) {
DEF_OP(Neg) {
auto Op = IROp->C<IR::IROp_Neg>();
if (Op->Cond == IR::CondClass::AL) {
if (Op->Cond == FEXCore::IR::COND_AL) {
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src));
} else {
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src), MapCC(Op->Cond));
@@ -1046,19 +1046,24 @@ DEF_OP(Popcount) {
if (CTX->HostFeatures.SupportsCSSC) {
switch (OpSize) {
case IR::OpSize::i8Bit:
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i16Bit:
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i32Bit: cnt(ARMEmitter::Size::i32Bit, Dst, Src); break;
case IR::OpSize::i64Bit: cnt(ARMEmitter::Size::i64Bit, Dst, Src); break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
case IR::OpSize::i8Bit:
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i16Bit:
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i32Bit:
cnt(ARMEmitter::Size::i32Bit, Dst, Src);
break;
case IR::OpSize::i64Bit:
cnt(ARMEmitter::Size::i64Bit, Dst, Src);
break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
}
} else {
}
else {
switch (OpSize) {
case IR::OpSize::i8Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
@@ -1190,19 +1195,6 @@ DEF_OP(Rev) {
}
}
DEF_OP(Rbit) {
auto Op = IROp->C<IR::IROp_Rbit>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = ConvertSize48(IROp);
const auto Dst = GetReg(Node);
const auto Src = GetReg(Op->Src);
rbit(EmitSize, Dst, Src);
}
DEF_OP(Bfi) {
auto Op = IROp->C<IR::IROp_Bfi>();
const auto EmitSize = ConvertSize(IROp);
@@ -1286,16 +1278,6 @@ DEF_OP(Sbfe) {
sbfx(ConvertSize(IROp), Dst, Src, Op->lsb, Op->Width);
}
DEF_OP(MaskGenerateFromBitWidth) {
auto Op = IROp->C<IR::IROp_MaskGenerateFromBitWidth>();
auto BitWidth = GetReg(Op->BitWidth);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1);
cmp(ARMEmitter::Size::i64Bit, BitWidth, 0);
lslv(ARMEmitter::Size::i64Bit, TMP2, TMP1, BitWidth);
csinv(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1, TMP2, ARMEmitter::Condition::CC_EQ);
}
DEF_OP(Select) {
auto Op = IROp->C<IR::IROp_Select>();
const auto OpSize = IROp->Size;
@@ -81,32 +81,35 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
Relocations.emplace_back(MoveABI);
}
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation> Relocations) {
const auto OrigBase = GetBufferBase();
const auto OrigSize = GetBufferSize();
const auto OrigOffset = GetCursorOffset();
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
const char* EntryRelocations) {
size_t DataIndex {};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
SetBuffer(reinterpret_cast<std::uint8_t*>(Code.data()), Code.size_bytes());
for (auto& Reloc : Relocations) {
switch (Reloc.Header.Type) {
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc.NamedSymbolLiteral.Symbol);
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
SetCursorOffset(Reloc.NamedSymbolLiteral.Offset);
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
dc64(Pointer);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(Reloc.NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
@@ -114,27 +117,18 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Co
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
SetBuffer(OrigBase, OrigSize);
SetCursorOffset(OrigOffset);
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(Reloc.GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIPMove.RegisterIndex), Pointer, true);
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
}
}
SetBuffer(OrigBase, OrigSize);
SetCursorOffset(OrigOffset);
return true;
}
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations() {
return std::move(Relocations);
}
} // namespace FEXCore::CPU
@@ -235,16 +235,16 @@ DEF_OP(CondJump) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1), "CondJump: Expected GPR");
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
if (Op->Cond == IR::CondClass::EQ) {
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
cbz(Size, Reg, TrueTargetLabel);
} else if (Op->Cond == IR::CondClass::NEQ) {
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
cbnz(Size, Reg, TrueTargetLabel);
} else if (Op->Cond == IR::CondClass::TSTZ) {
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
tbz(Reg, Const, TrueTargetLabel);
} else if (Op->Cond == IR::CondClass::TSTNZ) {
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
tbnz(Reg, Const, TrueTargetLabel);
} else {
@@ -423,11 +423,11 @@ DEF_OP(Vector_FToI) {
const auto Mask = PRED_TMP_32B.Merging();
switch (Op->Round) {
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
}
} else {
const auto IsScalar = ElementSize == OpSize;
@@ -449,21 +449,21 @@ DEF_OP(Vector_FToI) {
}
switch (Op->Round) {
case IR::RoundMode::Nearest: ROUNDING_FN(frintn); break;
case IR::RoundMode::NegInfinity: ROUNDING_FN(frintm); break;
case IR::RoundMode::PosInfinity: ROUNDING_FN(frintp); break;
case IR::RoundMode::TowardsZero: ROUNDING_FN(frintz); break;
case IR::RoundMode::Host: ROUNDING_FN(frinti); break;
case IR::Round_Nearest.Val: ROUNDING_FN(frintn); break;
case IR::Round_Negative_Infinity.Val: ROUNDING_FN(frintm); break;
case IR::Round_Positive_Infinity.Val: ROUNDING_FN(frintp); break;
case IR::Round_Towards_Zero.Val: ROUNDING_FN(frintz); break;
case IR::Round_Host.Val: ROUNDING_FN(frinti); break;
}
#undef ROUNDING_FN
} else {
switch (Op->Round) {
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
}
}
}
@@ -539,11 +539,11 @@ DEF_OP(Vector_F64ToI32) {
// Then convert to integers using fcvtzs.
auto CVTReg = Dst.Z();
switch (Round) {
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::RoundMode::TowardsZero: CVTReg = Vector.Z(); break;
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
case IR::Round_Towards_Zero.Val: CVTReg = Vector.Z(); break;
case IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
}
fcvtzs(Dst.Z(), ARMEmitter::SubRegSize::i32Bit, Mask, CVTReg, ARMEmitter::SubRegSize::i64Bit);
@@ -567,11 +567,11 @@ DEF_OP(Vector_F64ToI32) {
///< Round float to integral depending on rounding mode.
switch (Round) {
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::TowardsZero: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
}
// Now narrow from f64 to f32.
@@ -1,35 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/fextl/vector.h>
#include <cstdint>
namespace FEXCore::CPU {
union Relocation;
} // namespace FEXCore::CPU
namespace FEXCore::Core {
struct DebugDataSubblock {
uint32_t HostCodeOffset;
uint32_t HostCodeSize;
};
struct DebugDataGuestOpcode {
uint64_t GuestEntryOffset;
ptrdiff_t HostEntryOffset;
};
/**
* @brief Contains debug data for a block of code for later debugger analysis
*
* Needs to remain around for as long as the code could be executed at least
*/
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
fextl::vector<DebugDataSubblock> Subblocks;
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
};
} // namespace FEXCore::Core
+48 -57
View File
@@ -12,13 +12,14 @@ $end_info$
*/
#include "Common/SoftFloat.h"
#include "FEXCore/Utils/Telemetry.h"
#include "FEXCore/Utils/TypeDefines.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/JIT/DebugData.h"
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Utils/MemberFunctionToPointer.h"
@@ -29,16 +30,15 @@ $end_info$
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <cstdio>
#include <cstring>
#include <limits>
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include <stdio.h>
#include <unistd.h>
#include <string.h>
#include <limits>
namespace {
struct DivRem {
@@ -538,8 +538,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
} else {
{
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
auto lk_inval =
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
HostCode = Thread->LookupCache->FindBlock(GuestRip);
}
if (!HostCode) {
@@ -625,10 +624,10 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
RAPass->AddRegisters(IR::RegClass::GPR, GeneralRegisters.size());
RAPass->AddRegisters(IR::RegClass::GPRFixed, StaticRegisters.size());
RAPass->AddRegisters(IR::RegClass::FPR, GeneralFPRegisters.size());
RAPass->AddRegisters(IR::RegClass::FPRFixed, StaticFPRegisters.size());
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
RAPass->PairRegs = PairRegisters;
{
@@ -741,48 +740,48 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
}
}
void Arm64JITCore::EmitTFCheck() {
ARMEmitter::ForwardLabel l_TFUnset;
ARMEmitter::ForwardLabel l_TFBlocked;
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
if (CheckTF) {
ARMEmitter::ForwardLabel l_TFUnset;
ARMEmitter::ForwardLabel l_TFBlocked;
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
tbz(TMP1, 1, &l_TFBlocked);
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
tbz(TMP1, 1, &l_TFBlocked);
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
.FaultToTopAndGeneratedException = 1,
.Signal = Core::FAULT_SIGTRAP,
.TrapNo = X86State::X86_TRAPNO_DB,
.si_code = 2,
.err_code = 0,
};
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
.FaultToTopAndGeneratedException = 1,
.Signal = Core::FAULT_SIGTRAP,
.TrapNo = X86State::X86_TRAPNO_DB,
.si_code = 2,
.err_code = 0,
};
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
Bind(&l_TFBlocked);
// If TF was blocked for this instruction, unblock it for the next.
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
Bind(&l_TFUnset);
}
Bind(&l_TFBlocked);
// If TF was blocked for this instruction, unblock it for the next.
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
Bind(&l_TFUnset);
}
void Arm64JITCore::EmitSuspendInterruptCheck() {
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
// Trigger a fault if there are any pending interrupts
// Used only for suspend on WIN32 at the moment
@@ -807,9 +806,7 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
adr(TMP1, &HeaderLabel);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
if (CheckTF) {
EmitTFCheck();
}
EmitInterruptChecks(CheckTF);
if (SpillSlots) {
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
@@ -895,9 +892,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// if there's a pending branch, and it is not fall-through
if (PendingTargetLabel && PendingTargetLabel != Target) {
if (PendingTargetLabel->Backward.Location) {
EmitSuspendInterruptCheck();
}
b(PendingTargetLabel);
PendingTargetLabel = nullptr;
}
@@ -953,9 +947,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Make sure last branch is generated. It certainly can't be eliminated here.
if (PendingTargetLabel) {
if (PendingTargetLabel->Backward.Location) {
EmitSuspendInterruptCheck();
}
b(PendingTargetLabel);
}
PendingTargetLabel = nullptr;
+56 -76
View File
@@ -10,17 +10,13 @@ $end_info$
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/JIT/Relocations.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
@@ -29,19 +25,16 @@ $end_info$
#include <array>
#include <cstdint>
#include <functional>
#include <optional>
#include <utility>
#include <variant>
namespace FEXCore::Core {
struct InternalThreadState;
}
namespace FEXCore::Context {
struct ExitFunctionLinkData;
}
namespace FEXCore::IR {
class RegisterAllocationPass;
}
namespace FEXCore::CPU {
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
@@ -97,13 +90,11 @@ private:
[[nodiscard]]
ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const {
const auto RegClass = Reg.AsRegClass();
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::GPRFixed || RegClass == IR::RegClass::GPR, "Unexpected Class: {}", Reg.Class);
if (RegClass == IR::RegClass::GPRFixed) {
if (Reg.Class == IR::GPRFixedClass.Val) {
return StaticRegisters[Reg.Reg];
} else if (RegClass == IR::RegClass::GPR) {
} else if (Reg.Class == IR::GPRClass.Val) {
return GeneralRegisters[Reg.Reg];
}
@@ -122,13 +113,11 @@ private:
[[nodiscard]]
ARMEmitter::VRegister GetVReg(IR::PhysicalRegister Reg) const {
const auto RegClass = Reg.AsRegClass();
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::FPRFixed || RegClass == IR::RegClass::FPR, "Unexpected Class: {}", Reg.Class);
if (RegClass == IR::RegClass::FPRFixed) {
if (Reg.Class == IR::FPRFixedClass.Val) {
return StaticFPRegisters[Reg.Reg];
} else if (RegClass == IR::RegClass::FPR) {
} else if (Reg.Class == IR::FPRClass.Val) {
return GeneralFPRegisters[Reg.Reg];
}
@@ -146,8 +135,8 @@ private:
}
[[nodiscard]]
static IR::RegClass GetRegClass(IR::Ref Node) {
return IR::PhysicalRegister(Node).AsRegClass();
FEXCore::IR::RegisterClassType GetRegClass(IR::Ref Node) const {
return FEXCore::IR::RegisterClassType {IR::PhysicalRegister(Node).Class};
}
[[nodiscard]]
@@ -164,7 +153,7 @@ private:
// Converts IR-base shift type to ARMEmitter shift type.
// Will be a no-op, only a type conversion since the two definitions match.
[[nodiscard]]
static ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) {
ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
@@ -172,23 +161,18 @@ private:
}
[[nodiscard]]
static ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
return Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
}
[[nodiscard]]
static ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
return ConvertSize(Op);
}
[[nodiscard]]
static ARMEmitter::Size ConvertSize(IR::OpSize Size) {
return Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
"Invalid size");
@@ -200,105 +184,105 @@ private:
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
return ConvertSubRegSize16(Op->ElementSize);
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
LOGMAN_THROW_A_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
return ConvertSubRegSize16(ElementSize);
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
return ConvertSubRegSize8(Op->ElementSize);
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
return ConvertSubRegSize8(Op);
}
[[nodiscard]]
static ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
return ConvertSubRegSize8(Op);
}
[[nodiscard]]
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
return ARMEmitter::ToVectorSizePair(ConvertSubRegSize16(Op));
}
[[nodiscard]]
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
return ConvertSubRegSizePair16(Op);
}
[[nodiscard]]
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
return ConvertSubRegSizePair8(Op);
}
[[nodiscard]]
static ARMEmitter::Condition MapCC(IR::CondClass Cond) {
switch (Cond) {
case IR::CondClass::EQ: return ARMEmitter::Condition::CC_EQ;
case IR::CondClass::NEQ: return ARMEmitter::Condition::CC_NE;
case IR::CondClass::SGE: return ARMEmitter::Condition::CC_GE;
case IR::CondClass::SLT: return ARMEmitter::Condition::CC_LT;
case IR::CondClass::SGT: return ARMEmitter::Condition::CC_GT;
case IR::CondClass::SLE: return ARMEmitter::Condition::CC_LE;
case IR::CondClass::UGE: return ARMEmitter::Condition::CC_CS;
case IR::CondClass::ULT: return ARMEmitter::Condition::CC_CC;
case IR::CondClass::UGT: return ARMEmitter::Condition::CC_HI;
case IR::CondClass::ULE: return ARMEmitter::Condition::CC_LS;
case IR::CondClass::FLU: return ARMEmitter::Condition::CC_LT;
case IR::CondClass::FGE: return ARMEmitter::Condition::CC_GE;
case IR::CondClass::FLEU: return ARMEmitter::Condition::CC_LE;
case IR::CondClass::FGT: return ARMEmitter::Condition::CC_GT;
case IR::CondClass::FU:
case IR::CondClass::VS: return ARMEmitter::Condition::CC_VS;
case IR::CondClass::FNU:
case IR::CondClass::VC: return ARMEmitter::Condition::CC_VC;
case IR::CondClass::MI: return ARMEmitter::Condition::CC_MI;
case IR::CondClass::PL: return ARMEmitter::Condition::CC_PL;
ARMEmitter::Condition MapCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_FU:
case FEXCore::IR::COND_VS: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FNU:
case FEXCore::IR::COND_VC: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
}
}
[[nodiscard]]
static bool IsFPR(IR::RegClass Class) {
return Class == IR::RegClass::FPR || Class == IR::RegClass::FPRFixed;
bool IsFPR(IR::RegisterClassType Class) const {
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
}
[[nodiscard]]
static bool IsGPR(IR::RegClass Class) {
return Class == IR::RegClass::GPR || Class == IR::RegClass::GPRFixed;
bool IsGPR(IR::RegisterClassType Class) const {
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
}
[[nodiscard]]
static bool IsGPR(IR::Ref Node) {
bool IsGPR(IR::Ref Node) {
return IsGPR(GetRegClass(Node));
}
[[nodiscard]]
static bool IsFPR(IR::Ref Node) {
bool IsFPR(IR::Ref Node) {
return IsFPR(GetRegClass(Node));
}
[[nodiscard]]
static bool IsGPR(IR::OrderedNodeWrapper Wrap) {
return IsGPR(IR::PhysicalRegister(Wrap).AsRegClass());
bool IsGPR(IR::OrderedNodeWrapper Wrap) {
return IsGPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
}
[[nodiscard]]
static bool IsFPR(IR::OrderedNodeWrapper Wrap) {
return IsFPR(IR::PhysicalRegister(Wrap).AsRegClass());
bool IsFPR(IR::OrderedNodeWrapper Wrap) {
return IsFPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
}
[[nodiscard]]
@@ -397,9 +381,7 @@ private:
fextl::vector<FEXCore::CPU::Relocation> Relocations;
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation>);
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() override;
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
/** @} */
@@ -422,11 +404,9 @@ private:
void Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize);
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
void EmitTFCheck();
void EmitSuspendInterruptCheck();
void EmitInterruptChecks(bool CheckTF);
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
+56 -79
View File
@@ -21,7 +21,7 @@ DEF_OP(LoadContext) {
const auto Op = IROp->C<IR::IROp_LoadContext>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
auto Dst = GetReg(Node);
switch (OpSize) {
@@ -52,7 +52,7 @@ DEF_OP(LoadContext) {
DEF_OP(LoadContextPair) {
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst1 = GetReg(Op->OutValue1);
const auto Dst2 = GetReg(Op->OutValue2);
@@ -78,7 +78,7 @@ DEF_OP(StoreContext) {
const auto Op = IROp->C<IR::IROp_StoreContext>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
auto Src = GetZeroableReg(Op->Value);
switch (OpSize) {
@@ -110,7 +110,7 @@ DEF_OP(StoreContextPair) {
const auto Op = IROp->C<IR::IROp_StoreContextPair>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
auto Src1 = GetZeroableReg(Op->Value1);
auto Src2 = GetZeroableReg(Op->Value2);
@@ -135,11 +135,11 @@ DEF_OP(StoreContextPair) {
DEF_OP(LoadRegister) {
const auto Op = IROp->C<IR::IROp_LoadRegister>();
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == IR::GPRClass) {
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
} else if (Op->Class == IR::RegClass::FPR) {
} else if (Op->Class == IR::FPRClass) {
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
@@ -175,13 +175,12 @@ DEF_OP(LoadAF) {
DEF_OP(StoreRegister) {
const auto Op = IROp->C<IR::IROp_StoreRegister>();
const auto Reg = IR::PhysicalRegister(Node);
const auto RegClass = Reg.AsRegClass();
auto Reg = IR::PhysicalRegister(Node);
if (RegClass == IR::RegClass::GPRFixed) {
if (Reg.Class == IR::GPRFixedClass) {
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
mov(ARMEmitter::Size::i64Bit, GetReg(Reg), GetReg(Op->Value));
} else if (RegClass == IR::RegClass::FPRFixed) {
} else if (Reg.Class == IR::FPRFixedClass) {
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
@@ -194,7 +193,7 @@ DEF_OP(StoreRegister) {
mov(guest.Q(), host.Q());
}
} else {
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", RegClass);
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Reg.Class);
}
}
@@ -226,7 +225,7 @@ DEF_OP(LoadContextIndexed) {
const auto Index = GetReg(Op->Index);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
switch (Op->Stride) {
case 1:
case 2:
@@ -289,7 +288,7 @@ DEF_OP(StoreContextIndexed) {
const auto Index = GetReg(Op->Index);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Value = GetReg(Op->Value);
switch (Op->Stride) {
@@ -349,31 +348,12 @@ DEF_OP(StoreContextIndexed) {
}
}
DEF_OP(FormContextAddress) {
const auto Op = IROp->C<IR::IROp_FormContextAddress>();
const auto Index = GetReg(Op->Index);
const auto Dst = GetReg(Node);
switch (Op->Stride) {
case 1:
case 2:
case 4:
case 8:
case 16:
case 32: {
add(ARMEmitter::Size::i64Bit, Dst, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled FormContextAddress stride: {}", Op->Stride); break;
}
}
DEF_OP(SpillRegister) {
const auto Op = IROp->C<IR::IROp_SpillRegister>();
const auto OpSize = IROp->Size;
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetReg(Op->Value);
switch (OpSize) {
case IR::OpSize::i8Bit: {
@@ -414,7 +394,7 @@ DEF_OP(SpillRegister) {
}
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
}
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
} else if (Op->Class == FEXCore::IR::FPRClass) {
const auto Src = GetVReg(Op->Value);
switch (OpSize) {
@@ -453,7 +433,7 @@ DEF_OP(SpillRegister) {
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
}
} else {
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class);
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
}
}
@@ -462,7 +442,7 @@ DEF_OP(FillRegister) {
const auto OpSize = IROp->Size;
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
switch (OpSize) {
case IR::OpSize::i8Bit: {
@@ -503,7 +483,7 @@ DEF_OP(FillRegister) {
}
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
}
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
} else if (Op->Class == FEXCore::IR::FPRClass) {
const auto Dst = GetVReg(Node);
switch (OpSize) {
@@ -542,7 +522,7 @@ DEF_OP(FillRegister) {
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
}
} else {
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class);
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
}
}
@@ -579,14 +559,14 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, Const);
} else {
auto RegOffset = GetReg(Offset);
switch (OffsetType) {
case IR::MemOffsetType::SXTX:
switch (OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val:
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
case IR::MemOffsetType::UXTW:
case IR::MEM_OFFSET_UXTW.Val:
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
case IR::MemOffsetType::SXTW:
case IR::MEM_OFFSET_SXTW.Val:
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType); break;
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
}
}
}
@@ -613,20 +593,20 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
} else {
auto RegOffset = GetReg(Offset);
switch (OffsetType) {
case IR::MemOffsetType::SXTX:
switch (OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val:
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
break;
case IR::MemOffsetType::UXTW:
case IR::MEM_OFFSET_UXTW.Val:
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
break;
case IR::MemOffsetType::SXTW:
case IR::MEM_OFFSET_SXTW.Val:
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
break;
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType); break;
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType.Val); break;
}
}
return Tmp;
@@ -677,7 +657,7 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
// Note that we do nothing with the offset type and offset scale,
// since SVE loads and stores don't have the ability to perform an
// optional extension or shift as part of their behavior.
LOGMAN_THROW_A_FMT(OffsetType == IR::MemOffsetType::SXTX, "Currently only the default offset type (SXTX) is supported.");
LOGMAN_THROW_A_FMT(OffsetType.Val == IR::MEM_OFFSET_SXTX.Val, "Currently only the default offset type (SXTX) is supported.");
const auto RegOffset = GetReg(Offset);
return ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
@@ -690,7 +670,7 @@ DEF_OP(LoadMem) {
const auto MemReg = GetReg(Op->Addr);
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
switch (OpSize) {
@@ -724,7 +704,7 @@ DEF_OP(LoadMemPair) {
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
const auto Addr = GetReg(Op->Addr);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst1 = GetReg(Op->OutValue1);
const auto Dst2 = GetReg(Op->OutValue2);
@@ -752,13 +732,13 @@ DEF_OP(LoadMemTSO) {
const auto MemReg = GetReg(Op->Addr);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
}
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
@@ -780,7 +760,7 @@ DEF_OP(LoadMemTSO) {
// Half-barrier once back-patched.
nop();
}
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == IR::RegClass::GPR) {
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
@@ -795,7 +775,7 @@ DEF_OP(LoadMemTSO) {
// Half-barrier once back-patched.
nop();
}
} else if (Op->Class == IR::RegClass::GPR) {
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
@@ -1040,7 +1020,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
ARMEmitter::VRegister IncomingDst, std::optional<ARMEmitter::Register> BaseAddr,
ARMEmitter::VRegister VectorIndexLow, std::optional<ARMEmitter::VRegister> VectorIndexHigh,
ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize, size_t DataElementOffsetStart,
size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize) {
size_t IndexElementOffsetStart, uint8_t OffsetScale) {
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid element size");
const auto PerformSMove = [this](IR::OpSize ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
@@ -1116,17 +1096,17 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
// Calculate memory position for this gather load
if (BaseAddr.has_value()) {
if (VectorIndexSize == IR::OpSize::i32Bit) {
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
} else {
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
}
} else {
///< In this case we have no base address, All addresses come from the vector register itself
if (VectorIndexSize == IR::OpSize::i32Bit) {
// Sign extend and shift in to the 64-bit register
sbfiz(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
sbfiz(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
} else {
lsl(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
lsl(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
}
}
@@ -1184,8 +1164,7 @@ DEF_OP(VLoadVectorGatherMasked) {
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
const bool SupportsSVELoad = (HostSupportsSVE128 || HostSupportsSVE256) &&
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) &&
VectorIndexSize == IROp->ElementSize && Op->AddrSize == IR::OpSize::i64Bit;
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) && VectorIndexSize == IROp->ElementSize;
if (SupportsSVELoad) {
uint8_t SVEScale = FEXCore::ilog2(OffsetScale);
@@ -1243,7 +1222,7 @@ DEF_OP(VLoadVectorGatherMasked) {
} else {
LOGMAN_THROW_A_FMT(!Is256Bit, "Can't emulate this gather load in the backend! Programming error!");
Emulate128BitGather(IROp->Size, IROp->ElementSize, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale, Op->AddrSize);
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale);
}
}
@@ -1268,9 +1247,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
const bool SupportsSVELoad = HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4) && Op->AddrSize == IR::OpSize::i64Bit;
if (SupportsSVELoad) {
if (HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4)) {
ARMEmitter::SVEModType ModType = ARMEmitter::SVEModType::MOD_NONE;
if (OffsetScale != 1) {
ModType = ARMEmitter::SVEModType::MOD_LSL;
@@ -1324,7 +1301,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
}
} else {
Emulate128BitGather(IR::OpSize::i128Bit, IR::OpSize::i32Bit, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
IR::OpSize::i64Bit, 0, 0, OffsetScale, Op->AddrSize);
IR::OpSize::i64Bit, 0, 0, OffsetScale);
}
}
@@ -1625,7 +1602,7 @@ DEF_OP(StoreMem) {
const auto MemReg = GetReg(Op->Addr);
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
switch (OpSize) {
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
@@ -1736,7 +1713,7 @@ DEF_OP(StoreMemPair) {
const auto OpSize = IROp->Size;
const auto Addr = GetReg(Op->Addr);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src1 = GetZeroableReg(Op->Value1);
const auto Src2 = GetZeroableReg(Op->Value2);
switch (OpSize) {
@@ -1763,13 +1740,13 @@ DEF_OP(StoreMemTSO) {
const auto MemReg = GetReg(Op->Addr);
if (Op->Class == IR::RegClass::GPR) {
if (Op->Class == FEXCore::IR::GPRClass) {
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
}
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
@@ -1790,7 +1767,7 @@ DEF_OP(StoreMemTSO) {
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == IR::RegClass::GPR) {
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
if (OpSize == IR::OpSize::i8Bit) {
@@ -2303,7 +2280,7 @@ DEF_OP(ParanoidLoadMemTSO) {
auto MemReg = GetReg(Op->Addr);
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
@@ -2324,7 +2301,7 @@ DEF_OP(ParanoidLoadMemTSO) {
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == IR::RegClass::GPR) {
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (OpSize == IR::OpSize::i8Bit) {
@@ -2338,7 +2315,7 @@ DEF_OP(ParanoidLoadMemTSO) {
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == IR::RegClass::GPR) {
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
@@ -2391,7 +2368,7 @@ DEF_OP(ParanoidStoreMemTSO) {
auto MemReg = GetReg(Op->Addr);
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
@@ -2411,7 +2388,7 @@ DEF_OP(ParanoidStoreMemTSO) {
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == IR::RegClass::GPR) {
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
+9 -11
View File
@@ -10,12 +10,10 @@ $end_info$
#endif
#include "Interface/Context/Context.h"
#include "Interface/Core/JIT/DebugData.h"
#include "Interface/Core/JIT/JITClass.h"
#include "FEXCore/Debug/InternalThreadState.h"
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/EnumUtils.h>
namespace FEXCore::CPU {
@@ -48,10 +46,10 @@ DEF_OP(GuestOpcode) {
DEF_OP(Fence) {
auto Op = IROp->C<IR::IROp_Fence>();
switch (Op->Fence) {
case IR::FenceType::Load: dmb(ARMEmitter::BarrierScope::LD); break;
case IR::FenceType::LoadStore: dmb(ARMEmitter::BarrierScope::SY); break;
case IR::FenceType::Store: dmb(ARMEmitter::BarrierScope::ST); break;
case IR::FenceType::Inst: isb(); break;
case IR::Fence_Load.Val: dmb(ARMEmitter::BarrierScope::LD); break;
case IR::Fence_LoadStore.Val: dmb(ARMEmitter::BarrierScope::SY); break;
case IR::Fence_Store.Val: dmb(ARMEmitter::BarrierScope::ST); break;
case IR::Fence_Inst.Val: isb(); break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
}
}
@@ -108,10 +106,10 @@ DEF_OP(GetRoundingMode) {
// zero. Just swapping 01 and 10. That's a bitfield reverse. Round mode is in
// bottom two bits. After reversing as a 32-bit operation, it'll be in [31:30]
// and ripe for reinsertion back at 0.
static_assert(FEXCore::ToUnderlying(IR::RoundMode::Nearest) == 0);
static_assert(FEXCore::ToUnderlying(IR::RoundMode::NegInfinity) == 1);
static_assert(FEXCore::ToUnderlying(IR::RoundMode::PosInfinity) == 2);
static_assert(FEXCore::ToUnderlying(IR::RoundMode::TowardsZero) == 3);
static_assert(IR::ROUND_MODE_NEAREST == 0);
static_assert(IR::ROUND_MODE_NEGATIVE_INFINITY == 1);
static_assert(IR::ROUND_MODE_POSITIVE_INFINITY == 2);
static_assert(IR::ROUND_MODE_TOWARDS_ZERO == 3);
rbit(ARMEmitter::Size::i32Bit, TMP1, Dst);
bfi(ARMEmitter::Size::i64Bit, Dst, TMP1, 30, 2);
@@ -18,4 +18,18 @@ DEF_OP(RMWHandle) {
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(IROp->Args[0]));
}
DEF_OP(Swap1) {
auto Op = IROp->C<IR::IROp_Swap1>();
auto A = GetReg(Op->A), B = GetReg(Op->B);
LOGMAN_THROW_A_FMT(B == GetReg(Node), "Invariant");
mov(ARMEmitter::Size::i64Bit, TMP1, A);
mov(ARMEmitter::Size::i64Bit, A, B);
mov(ARMEmitter::Size::i64Bit, B, TMP1);
}
DEF_OP(Swap2) {
// Implemented above
}
} // namespace FEXCore::CPU
@@ -41,7 +41,6 @@ namespace FEXCore::CPU {
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
const auto OpSize = IROp->Size; \
const auto Is256Bit = OpSize == IR::OpSize::i256Bit; \
const auto Is128Bit = OpSize == IR::OpSize::i128Bit; \
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
\
const auto Dst = GetVReg(Node); \
@@ -50,10 +49,8 @@ namespace FEXCore::CPU {
\
if (HostSupportsSVE256 && Is256Bit) { \
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
} else if (Is128Bit) { \
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
} else { \
ARMOp(Dst.D(), Vector1.D(), Vector2.D()); \
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
} \
}
@@ -747,11 +744,11 @@ DEF_OP(VFToIScalarInsert) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
switch (RoundMode) {
case IR::RoundMode::Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
case IR::RoundMode::NegInfinity: frintm(SubRegSize.Scalar, Dst, Src); break;
case IR::RoundMode::PosInfinity: frintp(SubRegSize.Scalar, Dst, Src); break;
case IR::RoundMode::TowardsZero: frintz(SubRegSize.Scalar, Dst, Src); break;
case IR::RoundMode::Host: frinti(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Negative_Infinity: frintm(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Positive_Infinity: frintp(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Towards_Zero: frintz(SubRegSize.Scalar, Dst, Src); break;
case IR::Round_Host: frinti(SubRegSize.Scalar, Dst, Src); break;
}
};
File diff suppressed because it is too large. Load diff
+121 -203
View File
@@ -139,28 +139,27 @@ public:
FlushRegisterCache();
return _Jump(_TargetBlock);
}
IRPair<IROp_CondJump> CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClass _Cond = CondClass::NEQ,
IRPair<IROp_CondJump> CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClassType _Cond = {COND_NEQ},
IR::OpSize _CompareSize = OpSize::iInvalid) {
FlushRegisterCache();
return _CondJump(_Cmp1, _Cmp2, _TrueBlock, _FalseBlock, _Cond, _CompareSize);
}
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClass cond = CondClass::NEQ) {
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
FlushRegisterCache();
return _CondJump(ssa0, cond);
}
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClass cond = CondClass::NEQ) {
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
FlushRegisterCache();
return _CondJump(ssa0, ssa1, ssa2, cond);
}
IRPair<IROp_CondJump> CondJumpNZCV(CondClass Cond) {
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
FlushRegisterCache();
return _CondJump(InvalidNode, InvalidNode, InvalidNode, InvalidNode, Cond, OpSize::iInvalid, true);
}
IRPair<IROp_CondJump> CondJumpBit(Ref Src, unsigned Bit, bool Set) {
FlushRegisterCache();
auto InlineConst = _InlineConstant(Bit);
auto Cond = Set ? CondClass::TSTNZ : CondClass::TSTZ;
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, Cond, OpSize::iInvalid, false);
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, {Set ? COND_TSTNZ : COND_TSTZ}, OpSize::iInvalid, false);
}
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint = BranchHint::None) {
FlushRegisterCache();
@@ -210,17 +209,10 @@ public:
}
static bool CanHaveSideEffects(const FEXCore::X86Tables::X86InstInfo* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
if (TableInfo) {
if (TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
// If it is marked as having memory access then always say it has a side-effect.
// Not always true but better to be safe.
return true;
}
if (TableInfo->Flags & (X86Tables::InstFlags::FLAGS_SETS_RIP | X86Tables::InstFlags::FLAGS_BLOCK_END)) {
// Cooperative suspend interrupts can be triggered at any back-edge, the RIP must be reconstructed correctly in such cases
return true;
}
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
// If it is marked as having memory access then always say it has a side-effect.
// Not always true but better to be safe.
return true;
}
auto CanHaveSideEffects = false;
@@ -252,7 +244,7 @@ public:
auto ExitBlock = CreateNewCodeBlockAfter(BackwardBlock);
auto DF = GetRFLAG(X86State::RFLAG_DF_RAW_LOC);
CondJump(DF, Zero, ForwardBlock, BackwardBlock, CondClass::EQ);
CondJump(DF, Zero, ForwardBlock, BackwardBlock, {COND_EQ});
for (auto D = 0; D < 2; ++D) {
SetCurrentCodeBlock(D ? BackwardBlock : ForwardBlock);
@@ -301,8 +293,7 @@ public:
return ShouldDump;
}
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions,
bool Is64BitMode, bool MonoBackpatcherBlock);
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions, bool Is64BitMode, bool MonoBackpatcherBlock);
void Finalize();
// Dispatch builder functions
@@ -319,7 +310,6 @@ public:
void UnhandledOp(OpcodeArgs);
void MOVGPROp(OpcodeArgs, uint32_t SrcIndex);
void MOVGPRImmediate(OpcodeArgs);
void MOVGPRNTOp(OpcodeArgs);
void MOVVectorAlignedOp(OpcodeArgs);
void MOVVectorUnalignedOp(OpcodeArgs);
@@ -565,7 +555,7 @@ public:
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
RoundMode TranslateRoundType(uint8_t Mode);
RoundType TranslateRoundType(uint8_t Mode);
template<IR::OpSize ElementSize>
void InsertScalarRound(OpcodeArgs);
@@ -759,6 +749,7 @@ public:
void X87FXTRACT(OpcodeArgs);
void X87FYL2X(OpcodeArgs, bool IsFYL2XP1);
void X87LDENV(OpcodeArgs);
void X87LDSW(OpcodeArgs);
void X87ModifySTP(OpcodeArgs, bool Inc);
void X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2);
@@ -907,10 +898,6 @@ public:
void VPCLMULQDQOp(OpcodeArgs);
void CRC32(OpcodeArgs);
void Extrq_imm(OpcodeArgs);
void Insertq_imm(OpcodeArgs);
void Extrq(OpcodeArgs);
void Insertq(OpcodeArgs);
void BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDefinition);
void UnimplementedOp(OpcodeArgs);
@@ -989,6 +976,7 @@ public:
void AVX128_VPSIGN(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IROps IROp, IR::OpSize ElementSize);
Ref AVX128_VFCMPImpl(IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
void AVX128_VFCMP(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_MOVBetweenGPR_FPR(OpcodeArgs);
@@ -1007,7 +995,9 @@ public:
void AVX128_VINSERT(OpcodeArgs);
void AVX128_VINSERTPS(OpcodeArgs);
Ref AVX128_PHSUBImpl(Ref Src1, Ref Src2, size_t ElementSize);
void AVX128_VPHSUB(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_VPHSUBSW(OpcodeArgs);
void AVX128_VADDSUBP(OpcodeArgs, IR::OpSize ElementSize);
@@ -1099,8 +1089,8 @@ public:
void AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
void AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
RefPair AVX128_VPGatherQPSImpl(OpcodeArgs, Ref Dest, Ref Mask, RefVSIB VSIB);
RefPair AVX128_VPGatherImpl(OpcodeArgs, OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
RefPair AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB);
RefPair AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
void AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize);
@@ -1110,8 +1100,8 @@ public:
// End of AVX 128-bit implementation
// AVX 256-bit operations
void StoreResult_WithAVXInsert(VectorOpType Type, RegClass Class, FEXCore::X86Tables::DecodedOp Op, Ref Value,
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
void StoreResult_WithAVXInsert(VectorOpType Type, FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, Ref Value,
IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
if (Op->Dest.IsGPR() && Op->Dest.Data.GPR.GPR >= X86State::REG_XMM_0 && Op->Dest.Data.GPR.GPR <= X86State::REG_XMM_15 &&
GetGuestVectorLength() == OpSize::i256Bit && Type == VectorOpType::SSE) {
const auto gpr = Op->Dest.Data.GPR.GPR;
@@ -1162,7 +1152,7 @@ public:
}
}
void StoreContextHelper(IR::OpSize Size, RegClass Class, Ref Value, uint32_t Offset) {
void StoreContextHelper(IR::OpSize Size, RegisterClassType Class, Ref Value, uint32_t Offset) {
// For i128Bit, we won't see a normal Constant to inline, but as a special
// case we can replace with a 2x64-bit store which can use inline zeroes.
if (Size == OpSize::i128Bit) {
@@ -1174,7 +1164,7 @@ public:
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
Ref Zero = _Constant(0);
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, RegClass::GPR, Zero, Zero, Offset);
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Offset);
// XXX: This works around InlineConstant not having an associated
// register class, else we'd just do InlineConstant above.
@@ -1230,16 +1220,16 @@ public:
if (Index >= GPR0Index && Index <= GPR15Index) {
Ref R = _StoreRegister(Value, GPRSize);
R->Reg = PhysicalRegister(RegClass::GPRFixed, Index - GPR0Index).Raw;
R->Reg = PhysicalRegister(GPRFixedClass, Index - GPR0Index).Raw;
} else if (Index == PFIndex) {
_StorePF(Value, GPRSize);
} else if (Index == AFIndex) {
_StoreAF(Value, GPRSize);
} else if (Index >= FPR0Index && Index <= FPR15Index) {
Ref R = _StoreRegister(Value, VectorSize);
R->Reg = PhysicalRegister(RegClass::FPRFixed, Index - FPR0Index).Raw;
R->Reg = PhysicalRegister(FPRFixedClass, Index - FPR0Index).Raw;
} else if (Index == DFIndex) {
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(Core::CPUState, flags[X86State::RFLAG_DF_RAW_LOC]));
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(Core::CPUState, flags[X86State::RFLAG_DF_RAW_LOC]));
} else {
bool Partial = RegCache.Partial & (1ull << Index);
auto Size = Partial ? OpSize::i64Bit : CacheIndexToOpSize(Index);
@@ -1264,7 +1254,7 @@ public:
StoreContextHelper(Size, Class, Value, Offset);
// If Partial and MMX register, then we need to store all 1s in bits 64-80
if (Partial && Index >= MM0Index && Index <= MM7Index) {
_StoreContextGPR(OpSize::i16Bit, Constant(0xFFFF), Offset + 8);
_StoreContext(OpSize::i16Bit, IR::GPRClass, Constant(0xFFFF), Offset + 8);
}
}
}
@@ -1330,7 +1320,6 @@ protected:
private:
FEX_CONFIG_OPT(ReducedPrecisionMode, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(StrictReducedPrecisionMode, X87STRICTREDUCEDPRECISION);
struct JumpTargetInfo {
Ref BlockEntry;
@@ -1555,63 +1544,23 @@ private:
AddressMode DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, MemoryAccessType AccessType, bool IsLoad);
Ref LoadSource(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
Ref LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
const LoadSourceOptions& Options = {});
Ref LoadSourceGPR(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
const LoadSourceOptions& Options = {}) {
return LoadSource(RegClass::GPR, Op, Operand, Flags, Options);
}
Ref LoadSourceFPR(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
const LoadSourceOptions& Options = {}) {
return LoadSource(RegClass::FPR, Op, Operand, Flags, Options);
}
Ref LoadSource_WithOpSize(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize,
uint32_t Flags, const LoadSourceOptions& Options = {});
Ref LoadSourceGPR_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize, uint32_t Flags,
const LoadSourceOptions& Options = {}) {
return LoadSource_WithOpSize(RegClass::GPR, Op, Operand, OpSize, Flags, Options);
}
Ref LoadSourceFPR_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize, uint32_t Flags,
const LoadSourceOptions& Options = {}) {
return LoadSource_WithOpSize(RegClass::FPR, Op, Operand, OpSize, Flags, Options);
}
void StoreResult_WithOpSize(RegClass Class, X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResultGPR_WithOpSize(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult_WithOpSize(RegClass::GPR, Op, Operand, Src, OpSize, Align, AccessType);
}
void StoreResultFPR_WithOpSize(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult_WithOpSize(RegClass::FPR, Op, Operand, Src, OpSize, Align, AccessType);
}
void StoreResult(RegClass Class, X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align,
Ref LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
IR::OpSize OpSize, uint32_t Flags, const LoadSourceOptions& Options = {});
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op,
const FEXCore::X86Tables::DecodedOperand& Operand, const Ref Src, IR::OpSize OpSize, IR::OpSize Align,
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand,
const Ref Src, IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const Ref Src, IR::OpSize Align,
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResultGPR(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align = OpSize::iInvalid,
MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult(RegClass::GPR, Op, Operand, Src, Align, AccessType);
}
void StoreResultFPR(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align = OpSize::iInvalid,
MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult(RegClass::FPR, Op, Operand, Src, Align, AccessType);
}
void StoreResult(RegClass Class, X86Tables::DecodedOp Op, Ref Src, OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void StoreResultGPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align = OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult(RegClass::GPR, Op, Src, Align, AccessType);
}
void StoreResultFPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align = OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
StoreResult(RegClass::FPR, Op, Src, Align, AccessType);
}
// In several instances, it's desirable to get a base address with the segment offset
// applied to it. This pulls all the common-case appending into a single set of functions.
[[nodiscard]]
Ref MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize) {
Ref Mem = LoadSourceGPR_WithOpSize(Op, Operand, OpSize, Op->Flags, {.LoadData = false});
Ref Mem = LoadSource_WithOpSize(GPRClass, Op, Operand, OpSize, Op->Flags, {.LoadData = false});
return AppendSegmentOffset(Mem, Op->Flags);
}
[[nodiscard]]
@@ -1847,15 +1796,14 @@ private:
// For DF, we need to transform 0/1 into 1/-1
StoreDF(_SubShift(OpSize::i64Bit, Constant(1), Value, ShiftType::LSL, 1));
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
auto PackedTF = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
auto PackedTF = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
// unblocked bit before raising the exception.
auto NewPackedTF =
_Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
_StoreContextGPR(OpSize::i8Bit, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
auto NewPackedTF = _Select(FEXCore::IR::COND_EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
_StoreContext(OpSize::i8Bit, GPRClass, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
} else {
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
}
}
@@ -1881,12 +1829,12 @@ private:
}
[[nodiscard]]
static CondClass CondForNZCVBit(unsigned BitOffset, bool Invert) {
static CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
switch (BitOffset) {
case X86State::RFLAG_SF_RAW_LOC: return Invert ? CondClass::PL : CondClass::MI;
case X86State::RFLAG_ZF_RAW_LOC: return Invert ? CondClass::NEQ : CondClass::EQ;
case X86State::RFLAG_CF_RAW_LOC: return Invert ? CondClass::ULT : CondClass::UGE;
case X86State::RFLAG_OF_RAW_LOC: return Invert ? CondClass::FNU : CondClass::FU;
case X86State::RFLAG_SF_RAW_LOC: return {Invert ? COND_PL : COND_MI};
case X86State::RFLAG_ZF_RAW_LOC: return {Invert ? COND_NEQ : COND_EQ};
case X86State::RFLAG_CF_RAW_LOC: return {Invert ? COND_ULT : COND_UGE};
case X86State::RFLAG_OF_RAW_LOC: return {Invert ? COND_FNU : COND_FU};
default: FEX_UNREACHABLE;
}
}
@@ -1897,10 +1845,10 @@ private:
static const int PFIndex = 16;
static const int AFIndex = 17;
/* Gap 18..19 */
/* Note this range is only valid if MMXState = MMXState_MMX */
static const int MM0Index = 20;
static const int MM7Index = 27;
/* Gap 28..30 */
static const int AbridgedFTWIndex = 28;
/* Gap 29..30 */
static const int DFIndex = 31;
static const int FPR0Index = 32;
static const int FPR15Index = 47;
@@ -1912,16 +1860,17 @@ private:
switch (Index) {
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
default: return ~0U;
}
}
[[nodiscard]]
static RegClass CacheIndexClass(int Index) {
static RegisterClassType CacheIndexClass(int Index) {
if ((Index >= MM0Index && Index <= MM7Index) || Index >= FPR0Index) {
return RegClass::FPR;
return FPRClass;
} else {
return RegClass::GPR;
return GPRClass;
}
}
@@ -1953,14 +1902,14 @@ private:
RegCache.Written &= ~Bit;
}
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegClass Class, IR::OpSize Size) {
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
LOGMAN_THROW_A_FMT(Index < 64, "valid index");
uint64_t Bit = (1ull << (uint64_t)Index);
if (Size == OpSize::i128Bit && (RegCache.Partial & Bit)) {
// We need to load the full register extend if we previously did a partial access.
Ref Value = RegCache.Value[Index];
Ref Full = _LoadContext(Size, Class, Offset);
Ref Full = _LoadContext(Size, RegClass, Offset);
// If we did a partial store, we're inserting into the full register
if (RegCache.Written & Bit) {
@@ -1973,8 +1922,8 @@ private:
if (!(RegCache.Cached & Bit)) {
if (Index == DFIndex) {
RegCache.Value[Index] = _LoadDF();
} else if ((Index >= MM0Index && Index <= MM7Index) || Index >= AVXHigh0Index) {
RegCache.Value[Index] = _LoadContext(Size, Class, Offset);
} else if ((Index >= MM0Index && Index <= AbridgedFTWIndex) || Index >= AVXHigh0Index) {
RegCache.Value[Index] = _LoadContext(Size, RegClass, Offset);
// We may have done a partial load, this requires special handling.
if (Size == OpSize::i64Bit) {
@@ -1985,7 +1934,7 @@ private:
} else if (Index == AFIndex) {
RegCache.Value[Index] = _LoadAF(Size);
} else {
RegCache.Value[Index] = _LoadRegister(Offset, Class, Size);
RegCache.Value[Index] = _LoadRegister(Offset, RegClass, Size);
}
RegCache.Cached |= Bit;
@@ -1994,21 +1943,21 @@ private:
return RegCache.Value[Index];
}
RefPair AllocatePair(RegClass Class, IR::OpSize Size) {
if (Class == RegClass::FPR) {
RefPair AllocatePair(FEXCore::IR::RegisterClassType Class, IR::OpSize Size) {
if (Class == FPRClass) {
return {_AllocateFPR(Size, Size), _AllocateFPR(Size, Size)};
} else {
return {_AllocateGPR(false), _AllocateGPR(false)};
}
}
RefPair LoadContextPair_Uncached(RegClass Class, IR::OpSize Size, unsigned Offset) {
RefPair LoadContextPair_Uncached(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, unsigned Offset) {
RefPair Values = AllocatePair(Class, Size);
_LoadContextPair(Size, Class, Offset, Values.Low, Values.High);
return Values;
}
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegClass Class, IR::OpSize Size) {
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
LOGMAN_THROW_A_FMT(Index != DFIndex, "must be pairable");
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
@@ -2016,7 +1965,7 @@ private:
uint64_t Bits = (3ull << (uint64_t)Index);
const auto SizeInt = IR::OpSizeToSize(Size);
if (((RegCache.Partial | RegCache.Cached) & Bits) == 0 && ((Offset / SizeInt) < 64)) {
auto Values = LoadContextPair_Uncached(Class, Size, Offset);
auto Values = LoadContextPair_Uncached(RegClass, Size, Offset);
RegCache.Value[Index] = Values.Low;
RegCache.Value[Index + 1] = Values.High;
RegCache.Cached |= Bits;
@@ -2028,13 +1977,13 @@ private:
// Fallback on a pair of loads
return {
.Low = LoadRegCache(Offset, Index, Class, Size),
.High = LoadRegCache(Offset + SizeInt, Index + 1, Class, Size),
.Low = LoadRegCache(Offset, Index, RegClass, Size),
.High = LoadRegCache(Offset + SizeInt, Index + 1, RegClass, Size),
};
}
Ref LoadGPR(uint8_t Reg) {
return LoadRegCache(Reg, GPR0Index + Reg, RegClass::GPR, GetGPROpSize());
return LoadRegCache(Reg, GPR0Index + Reg, GPRClass, GetGPROpSize());
}
Ref LoadContext(IR::OpSize Size, uint8_t Index) {
@@ -2050,7 +1999,7 @@ private:
}
Ref LoadXMMRegister(uint8_t Reg) {
return LoadRegCache(Reg, FPR0Index + Reg, RegClass::FPR, GetGuestVectorLength());
return LoadRegCache(Reg, FPR0Index + Reg, FPRClass, GetGuestVectorLength());
}
Ref LoadDF() {
@@ -2105,7 +2054,7 @@ private:
// Recover the sign bit, it is the logical DF value
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63));
} else {
return _LoadContextGPR(OpSize::i8Bit, offsetof(Core::CPUState, flags[BitOffset]));
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(Core::CPUState, flags[BitOffset]));
}
}
@@ -2122,18 +2071,18 @@ private:
}
// Safe version of NZCVSelect that handles inverted carries automatically.
Ref NZCVSelect(OpSize OpSize, CondClass Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
Ref NZCVSelect(OpSize OpSize, CondClassType Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
switch (Cond) {
case CondClass::UGE: /* cs */
case CondClass::ULT: /* cc */
case IR::COND_UGE: /* cs */
case IR::COND_ULT: /* cc */
// Invert the condition to match our expectations.
if (CarryIsInverted != CFInverted) {
Cond = (Cond == CondClass::UGE) ? CondClass::ULT : CondClass::UGE;
Cond = {Cond == COND_UGE ? COND_ULT : COND_UGE};
}
break;
case CondClass::UGT: /* hi */
case CondClass::ULE: /* ls */
case IR::COND_UGT: /* hi */
case IR::COND_ULE: /* ls */
// No clever optimization we can do here, rectify carry itself.
RectifyCarryInvert(CarryIsInverted);
break;
@@ -2222,7 +2171,7 @@ private:
HandleNZCV_RMW();
CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF, CFInverted));
StoreResultGPR(Op, Result);
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
}
// Helper to derive Dest by a given builder-using Expression with the opcode
@@ -2295,7 +2244,8 @@ private:
CachedIndexedNamedVectorConstants.clear();
}
std::optional<CondClass> DecodeNZCVCondition(uint8_t OP);
std::optional<CondClassType> DecodeNZCVCondition(uint8_t OP);
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
Ref SelectCC0All1(uint8_t OP);
/**
@@ -2309,8 +2259,8 @@ private:
if (Size != OpSize::i32Bit) {
return;
}
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
StoreResultGPR(Op, Dest);
auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
}
using ZeroShiftFunctionPtr = void (OpDispatchBuilder::*)(FEXCore::X86Tables::DecodedOp Op);
@@ -2345,7 +2295,7 @@ private:
///< Jump to zeroshift block or end block depending on if it was provided.
IRPair<IROp_CodeBlock> TailHandling = ZeroShiftResult ? ZeroShiftBlock : EndBlock;
CondJump(Shift, Zero, TailHandling, SetBlock, CondClass::EQ);
CondJump(Shift, Zero, TailHandling, SetBlock, {COND_EQ});
SetCurrentCodeBlock(SetBlock);
StartNewBlock();
@@ -2388,7 +2338,9 @@ private:
void CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High);
void CalculateFlags_UMUL(Ref High);
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res);
void CalculateFlags_ShiftLeft(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
void CalculateFlags_ShiftRight(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
void CalculateFlags_ShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
void CalculateFlags_ShiftRightDoubleImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
void CalculateFlags_ShiftRightImmediateCommon(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
@@ -2404,8 +2356,8 @@ private:
void ChgStateX87_MMX() override {
LOGMAN_THROW_A_FMT(MMXState == MMXState_X87, "Expected state to be x87");
_StackForceSlow();
SetX87Top(Constant(0)); // top reset to zero
_StoreContextGPR(OpSize::i8Bit, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
SetX87Top(Constant(0)); // top reset to zero
StoreContext(AbridgedFTWIndex, Constant(0xFFFFUL)); // all valid
MMXState = MMXState_MMX;
}
@@ -2442,62 +2394,44 @@ private:
IROp_IRHeader* CurrentHeader {};
[[nodiscard]]
bool IsTSOEnabled(RegClass Class) const {
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) const {
if (ForceTSO == ForceTSOMode::ForceEnabled) {
return true;
} else if (ForceTSO == ForceTSOMode::ForceDisabled) {
return false;
} else if (Class == RegClass::FPR) {
} else if (Class == FPRClass) {
return CTX->IsVectorAtomicTSOEnabled();
} else {
return CTX->IsAtomicTSOEnabled();
}
}
Ref _StoreMemAutoTSO(RegClass Class, OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
if (IsTSOEnabled(Class)) {
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
} else {
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
}
Ref _StoreMemGPRAutoTSO(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMemAutoTSO(RegClass::GPR, Size, Addr, Value, Align);
}
Ref _StoreMemFPRAutoTSO(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMemAutoTSO(RegClass::FPR, Size, Addr, Value, Align);
}
Ref _LoadMemAutoTSO(RegClass Class, OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = IR::OpSize::i8Bit) {
if (IsTSOEnabled(Class)) {
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
} else {
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
}
Ref _LoadMemGPRAutoTSO(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
return _LoadMemAutoTSO(RegClass::GPR, Size, ssa0, Align);
}
Ref _LoadMemFPRAutoTSO(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
return _LoadMemAutoTSO(RegClass::FPR, Size, ssa0, Align);
}
Ref _LoadMemAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
const auto B = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != RegClass::GPR, Size);
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
if (AtomicTSO) {
return _LoadMemTSO(Class, Size, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
} else {
return _LoadMem(Class, Size, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
return _LoadMem(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
}
}
Ref _LoadMemGPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
return _LoadMemAutoTSO(RegClass::GPR, Size, A, Align);
}
Ref _LoadMemFPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
return _LoadMemAutoTSO(RegClass::FPR, Size, A, Align);
}
AddressMode SelectPairAddressMode(AddressMode A, IR::OpSize Size) {
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
@@ -2515,72 +2449,56 @@ private:
}
RefPair LoadMemPair(RegClass Class, OpSize Size, Ref Base, uint32_t Offset) {
RefPair LoadMemPair(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Base, unsigned Offset) {
RefPair Values = AllocatePair(Class, Size);
_LoadMemPair(Class, Size, Base, Offset, Values.Low, Values.High);
return Values;
}
RefPair LoadMemPairFPR(OpSize Size, Ref Base, uint32_t Offset) {
return LoadMemPair(RegClass::FPR, Size, Base, Offset);
}
RefPair _LoadMemPairAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
RefPair _LoadMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
// Use ldp if possible, otherwise fallback on two loads.
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
const auto B = SelectPairAddressMode(A, Size);
return LoadMemPair(Class, Size, B.Base, B.Offset);
A = SelectPairAddressMode(A, Size);
return LoadMemPair(Class, Size, A.Base, A.Offset);
} else {
AddressMode HighA = A;
HighA.Offset += 16;
return {
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
};
}
AddressMode HighA = A;
HighA.Offset += 16;
return {
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
};
}
RefPair _LoadMemPairFPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
return _LoadMemPairAutoTSO(RegClass::FPR, Size, A, Align);
}
Ref _StoreMemAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
const auto B = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != RegClass::GPR, Size);
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
if (AtomicTSO) {
return _StoreMemTSO(Class, Size, Value, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
} else {
return _StoreMem(Class, Size, Value, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
return _StoreMem(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
}
}
Ref _StoreMemGPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMemAutoTSO(RegClass::GPR, Size, A, Value, Align);
}
Ref _StoreMemFPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMemAutoTSO(RegClass::FPR, Size, A, Value, Align);
}
void _StoreMemPairAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, Ref Value1, Ref Value2, OpSize Align = OpSize::i8Bit) {
void _StoreMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value1, Ref Value2,
IR::OpSize Align = IR::OpSize::i8Bit) {
const auto SizeInt = IR::OpSizeToSize(Size);
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
// Use stp if possible, otherwise fallback on two stores.
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
const auto B = SelectPairAddressMode(A, Size);
_StoreMemPair(Class, Size, Value1, Value2, B.Base, B.Offset);
A = SelectPairAddressMode(A, Size);
_StoreMemPair(Class, Size, Value1, Value2, A.Base, A.Offset);
} else {
auto B = A;
_StoreMemAutoTSO(Class, Size, B, Value1, OpSize::i8Bit);
B.Offset += SizeInt;
_StoreMemAutoTSO(Class, Size, B, Value2, OpSize::i8Bit);
_StoreMemAutoTSO(Class, Size, A, Value1, OpSize::i8Bit);
A.Offset += SizeInt;
_StoreMemAutoTSO(Class, Size, A, Value2, OpSize::i8Bit);
}
}
void _StoreMemPairFPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value1, Ref Value2, OpSize Align = OpSize::i8Bit) {
return _StoreMemPairAutoTSO(RegClass::FPR, Size, A, Value1, Value2, Align);
}
Ref Pop(IR::OpSize Size, Ref SP_RMW) {
Ref Value = _AllocateGPR(false);
@@ -35,16 +35,20 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
} else {
LOGMAN_THROW_A_FMT(IsOperandMem(Operand, true), "only memory sources");
AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
AddressMode HighA = A;
HighA.Offset += 16;
if (Operand.IsSIB()) {
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
LOGMAN_THROW_A_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
}
const AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
if (NeedsHigh) {
return _LoadMemPairFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit);
return _LoadMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit);
} else {
return {.Low = _LoadMemFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit)};
return {.Low = _LoadMemAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit)};
}
}
}
@@ -91,9 +95,9 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
if (Src.High) {
_StoreMemPairFPRAutoTSO(OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
_StoreMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
} else {
_StoreMemFPRAutoTSO(OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
_StoreMemAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
}
}
}
@@ -147,13 +151,13 @@ void OpDispatchBuilder::AVX128_VMOVScalarImpl(OpcodeArgs, IR::OpSize ElementSize
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Result, .High = High});
} else if (Op->Dest.IsGPR()) {
// VMOVSS/SD xmm1, mem32/mem64
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags);
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], ElementSize, Op->Flags);
auto High = LoadZeroVector(OpSize::i128Bit);
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Src, .High = High});
} else {
// VMOVSS/SD mem32/mem64, xmm1
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, ElementSize);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, ElementSize, OpSize::iInvalid);
}
}
@@ -347,7 +351,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
if (Op->Dest.IsGPR()) {
///< MOVNTDQA load non-temporal comes from SSE4.1 and is extended by AVX/AVX2.
RefPair Src {};
Ref SrcAddr = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
Ref SrcAddr = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
Src.Low = _VLoadNonTemporal(OpSize::i128Bit, SrcAddr, 0);
if (Is128Bit) {
@@ -358,7 +362,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
} else {
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit, MemoryAccessType::STREAM);
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
if (Is128Bit) {
// Single store non-temporal for 128-bit operations.
@@ -375,7 +379,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
if (Op->Src[0].IsGPR()) {
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
} else {
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
}
// This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 256bit
@@ -386,7 +390,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
Src.High = ZeroVector;
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
} else {
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
}
}
@@ -395,7 +399,7 @@ void OpDispatchBuilder::AVX128_VMOVLP(OpcodeArgs) {
if (!Op->Dest.IsGPR()) {
///< VMOVLPS/PD mem64, xmm1
StoreResultFPR_WithOpSize(Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
} else if (!Op->Src[1].IsGPR()) {
///< VMOVLPS/PD xmm1, xmm2, mem64
// Bits[63:0] come from Src2[63:0]
@@ -459,7 +463,7 @@ void OpDispatchBuilder::AVX128_VMOVDDUP(OpcodeArgs) {
// 128-bit operation only loads 8-bytes.
// 256-bit operation loads a full 32-bytes.
if (Is128Bit) {
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
} else {
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, true);
}
@@ -554,18 +558,18 @@ void OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstEle
if (Op->Src[1].IsGPR()) {
// If the source is a GPR then convert directly from the GPR.
auto Src2 = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GetGPROpSize(), Op->Flags);
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], GetGPROpSize(), Op->Flags);
Result.Low = _VSToFGPRInsert(OpSize::i128Bit, DstElementSize, SrcSize, Src1.Low, Src2, false);
} else if (SrcSize != DstElementSize) {
// If the source is from memory but the Source size and destination size aren't the same,
// then it is more optimal to load in to a GPR and convert between GPR->FPR.
// ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't.
auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags);
auto Src2 = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags);
Result.Low = _VSToFGPRInsert(DstSize, DstElementSize, SrcSize, Src1.Low, Src2, false);
} else {
// In the case of cvtsi2s{s,d} where the source and destination are the same size,
// then it is more optimal to load in to the FPR register directly and convert there.
auto Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
auto Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
// Always signed
Result.Low = _VSToFVectorInsert(DstSize, DstElementSize, DstElementSize, Src1.Low, Src2, false, false);
}
@@ -585,11 +589,11 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi
if (Op->Src[0].IsGPR()) {
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
} else {
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcElementSize, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcElementSize, Op->Flags);
}
Ref Result = CVTFPR_To_GPRImpl(Op, Src.Low, SrcElementSize, HostRoundingMode);
StoreResultGPR(Op, Result);
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
@@ -632,7 +636,7 @@ void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize) {
if (Op->Src[0].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
} else {
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
}
Comiss(ElementSize, Src1.Low, Src2.Low);
@@ -649,7 +653,7 @@ void OpDispatchBuilder::AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IR
if (Op->Src[1].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
} else {
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
}
// If OpSize == ElementSize then it only does the lower scalar op
@@ -686,7 +690,7 @@ void OpDispatchBuilder::AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSi
if (Op->Src[1].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
} else {
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
}
const uint8_t CompType = Op->Src[2].Literal();
@@ -704,12 +708,12 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
RefPair Result {};
if (Op->Src[0].IsGPR()) {
// Loading from GPR and moving to Vector.
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], GetGPROpSize(), Op->Flags);
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], GetGPROpSize(), Op->Flags);
// zext to 128bit
Result.Low = _VCastFromGPR(OpSize::i128Bit, OpSizeFromSrc(Op), Src);
} else {
// Loading from Memory as a scalar. Zero extend
Result.Low = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Result.Low = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
}
Result.High = LoadZeroVector(OpSize::i128Bit);
@@ -722,11 +726,11 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
auto ElementSize = OpSizeFromDst(Op);
// Extract element from GPR. Zero extending in the process.
Src.Low = _VExtractToGPR(OpSizeFromSrc(Op), ElementSize, Src.Low, 0);
StoreResultGPR(Op, Op->Dest, Src.Low);
StoreResult(GPRClass, Op, Op->Dest, Src.Low, OpSize::iInvalid);
} else {
// Storing first element to memory.
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
_StoreMemFPR(OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
_StoreMem(FPRClass, OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
}
}
}
@@ -754,7 +758,7 @@ void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs, IR::OpSize ElementSize) {
const auto GPRSize = GetGPROpSize();
// Extract already zero extends the result.
Ref Result = _VExtractToGPR(OpSize::i128Bit, OverridenElementSize, Src.Low, Index);
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, GPRSize);
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
return;
}
@@ -775,7 +779,7 @@ void OpDispatchBuilder::AVX128_ExtendVectorElements(OpcodeArgs, IR::OpSize Eleme
const auto SrcSize = OpSizeFromSrc(Op);
const auto LoadSize = Is256Bit ? IR::SizeToOpSize(IR::OpSizeToSize(SrcSize) * 2) : SrcSize;
return LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
return LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
}
};
@@ -864,7 +868,7 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
auto GPRHigh = Mask8Byte(Src.High);
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 2);
}
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid);
}
void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
@@ -893,7 +897,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
Result = _Orlshl(OpSize::i64Bit, Result, ResultHigh, 16);
}
StoreResultGPR(Op, Result);
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, const X86Tables::DecodedOperand& Src1Op,
@@ -906,7 +910,7 @@ void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, con
if (Src2Op.IsGPR()) {
// If the source is a GPR then convert directly from the GPR.
auto Src2 = LoadSourceGPR_WithOpSize(Op, Src2Op, GetGPROpSize(), Op->Flags);
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, GetGPROpSize(), Op->Flags);
Result.Low = _VInsGPR(OpSize::i128Bit, ElementSize, Index, Src1.Low, Src2);
} else {
// If loading from memory then we only load the element size
@@ -1043,7 +1047,7 @@ void OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::O
// Then zero extends the top 128-bit.
const auto SrcSize = Op->Src[1].IsGPR() ? OpSize::i128Bit : SrcElementSize;
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
Ref Result = _VFToFScalarInsert(OpSize::i128Bit, DstElementSize, SrcElementSize, Src1.Low, Src2, false);
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result));
@@ -1072,7 +1076,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize
} else {
// Handle 64-bit memory source.
// In the case of cvtps2pd xmm, m64.
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
}
RefPair Result {};
@@ -1150,7 +1154,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize Sr
// unnecessarily zero extend the vector. Otherwise, if
// memory, then we want to load the element size exactly.
const auto LoadSize = IR::SizeToOpSize(8 * (IR::OpSizeToSize(Size) / 16));
return RefPair {.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags)};
return RefPair {.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags)};
} else {
return AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
}
@@ -1300,7 +1304,7 @@ void OpDispatchBuilder::AVX128_InsertScalarRound(OpcodeArgs, IR::OpSize ElementS
if (Op->Src[1].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
} else {
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
}
// If OpSize == ElementSize then it only does the lower scalar op
@@ -1569,20 +1573,20 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
auto Address = MakeAddress(Op->Dest);
auto Data = AVX128_LoadSource_WithOpSize(Op, DataOp, Op->Flags, !Is128Bit);
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MemOffsetType::SXTX, 1);
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
if (!Is128Bit) {
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MemOffsetType::SXTX, 1);
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
}
} else {
auto Address = MakeAddress(DataOp);
RefPair Result {};
Result.Low = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Address, Invalid(), MemOffsetType::SXTX, 1);
Result.Low = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
if (Is128Bit) {
Result.High = LoadZeroVector(OpSize::i128Bit);
} else {
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MemOffsetType::SXTX, 1);
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
}
@@ -1612,11 +1616,11 @@ void OpDispatchBuilder::AVX128_MASKMOV(OpcodeArgs) {
// RDI source (DS prefix by default)
auto MemDest = MakeSegmentAddress(X86State::REG_RDI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
Ref XMMReg = _LoadMemFPR(Size, MemDest, OpSize::i8Bit);
Ref XMMReg = _LoadMem(FPRClass, Size, MemDest, OpSize::i8Bit);
// If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant
XMMReg = _VBSL(Size, MaskSrc.Low, VectorSrc.Low, XMMReg);
_StoreMemFPR(Size, MemDest, XMMReg, OpSize::i8Bit);
_StoreMem(FPRClass, Size, MemDest, XMMReg, OpSize::i8Bit);
}
void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) {
@@ -1656,7 +1660,7 @@ void OpDispatchBuilder::AVX128_SaveAVXState(Ref MemBase) {
for (uint32_t i = 0; i < NumRegs; i += 2) {
RefPair Pair = LoadContextPair(OpSize::i128Bit, AVXHigh0Index + i);
_StoreMemPairFPR(OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
_StoreMemPair(FPRClass, OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
}
}
@@ -1664,7 +1668,7 @@ void OpDispatchBuilder::AVX128_RestoreAVXState(Ref MemBase) {
const auto NumRegs = Is64BitMode ? 16U : 8U;
for (uint32_t i = 0; i < NumRegs; i += 2) {
auto YMMHRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 576);
auto YMMHRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 576);
AVX128_StoreXMMRegister(i, YMMHRegs.Low, true);
AVX128_StoreXMMRegister(i + 1, YMMHRegs.High, true);
@@ -1964,7 +1968,7 @@ void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Sr
if (Op->Src[1].IsGPR()) {
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false).Low;
} else {
Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
}
Ref Sources[3] = {Dest, Src1, Src2};
@@ -2012,8 +2016,8 @@ void OpDispatchBuilder::AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Sr
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
}
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize,
RefPair Dest, RefPair Mask, RefVSIB VSIB) {
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest,
RefPair Mask, RefVSIB VSIB) {
LOGMAN_THROW_A_FMT(AddrElementSize == OpSize::i32Bit || AddrElementSize == OpSize::i64Bit, "Unknown address element size");
const auto Is128Bit = Size == OpSize::i128Bit;
@@ -2057,13 +2061,10 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, Op
}
}
const auto GPRSize = GetGPROpSize();
auto AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
RefPair Result {};
///< Calculate the low-half.
Result.Low = _VLoadVectorGatherMasked(OpSize::i128Bit, ElementLoadSize, Dest.Low, Mask.Low, BaseAddr, VSIB.Low, VSIB.High,
AddrElementSize, VSIB.Scale, 0, 0, AddrSize);
AddrElementSize, VSIB.Scale, 0, 0);
if (Is128Bit) {
Result.High = LoadZeroVector(OpSize::i128Bit);
@@ -2100,7 +2101,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, Op
///< Calculate the high-half.
auto ResultHigh = _VLoadVectorGatherMasked(OpSize::i128Bit, ElementLoadSize, DestReg, MaskReg, BaseAddr, AddrAddressing.Low,
AddrAddressing.High, AddrElementSize, VSIB.Scale, DataElementOffset, IndexElementOffset, AddrSize);
AddrAddressing.High, AddrElementSize, VSIB.Scale, DataElementOffset, IndexElementOffset);
if (AddrElementSize == OpSize::i64Bit && ElementLoadSize == OpSize::i32Bit) {
// If we only fetched 128-bits worth of data then the upper-result is all zero.
@@ -2113,7 +2114,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, Op
return Result;
}
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(OpcodeArgs, Ref Dest, Ref Mask, RefVSIB VSIB) {
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB) {
///< BaseAddr doesn't need to exist, calculate that here.
Ref BaseAddr = VSIB.BaseAddr;
@@ -2141,11 +2142,8 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(OpcodeArgs,
RefPair Result {};
const auto GPRSize = GetGPROpSize();
auto AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
///< Calculate the low-half.
Result.Low = _VLoadVectorGatherMaskedQPS(OpSize::i128Bit, OpSize::i32Bit, Dest, Mask, BaseAddr, VSIB.Low, VSIB.High, VSIB.Scale, AddrSize);
Result.Low = _VLoadVectorGatherMaskedQPS(OpSize::i128Bit, OpSize::i32Bit, Dest, Mask, BaseAddr, VSIB.Low, VSIB.High, VSIB.Scale);
Result.High = LoadZeroVector(OpSize::i128Bit);
if (VSIB.High == Invalid()) {
// Special case for only loading two floats.
@@ -2204,15 +2202,15 @@ void OpDispatchBuilder::AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize) {
}
///< AddressElementSize is now OpSize::i64Bit
Result = AVX128_VPGatherQPSImpl(Op, Dest.Low, Mask.Low, VSIBLow);
Result = AVX128_VPGatherQPSImpl(Dest.Low, Mask.Low, VSIBLow);
if (NeedsHighAddrBytes) {
auto Res = AVX128_VPGatherQPSImpl(Op, Dest.High, Mask.High, VSIBHigh);
auto Res = AVX128_VPGatherQPSImpl(Dest.High, Mask.High, VSIBHigh);
Result.High = Res.Low;
}
} else if (AddrElementSize == OpSize::i64Bit && ElementLoadSize == OpSize::i32Bit) {
Result = AVX128_VPGatherQPSImpl(Op, Dest.Low, Mask.Low, VSIB);
Result = AVX128_VPGatherQPSImpl(Dest.Low, Mask.Low, VSIB);
} else {
Result = AVX128_VPGatherImpl(Op, Size, ElementLoadSize, AddrElementSize, Dest, Mask, VSIB);
Result = AVX128_VPGatherImpl(Size, ElementLoadSize, AddrElementSize, Dest, Mask, VSIB);
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
@@ -2236,7 +2234,7 @@ void OpDispatchBuilder::AVX128_VCVTPH2PS(OpcodeArgs) {
// In the event that a memory operand is used as the source operand,
// the access width will always be half the size of the destination vector width
// (i.e. 128-bit vector -> 64-bit mem, 256-bit vector -> 128-bit mem)
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
}
RefPair Result {};
@@ -2291,7 +2289,7 @@ void OpDispatchBuilder::AVX128_VCVTPS2PH(OpcodeArgs) {
}
if (!Op->Dest.IsGPR()) {
StoreResultFPR_WithOpSize(Op, Op->Dest, Result.Low, StoreSize);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result.Low, StoreSize, OpSize::iInvalid);
} else {
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
}
@@ -53,7 +53,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
{0xAA, 2, &OpDispatchBuilder::STOSOp},
{0xAC, 2, &OpDispatchBuilder::LODSOp},
{0xAE, 2, &OpDispatchBuilder::SCASOp},
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPRImmediate>},
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
{0xC2, 2, &OpDispatchBuilder::RETOp},
{0xC8, 1, &OpDispatchBuilder::EnterOp},
{0xC9, 1, &OpDispatchBuilder::LEAVEOp},
@@ -23,8 +23,8 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
@@ -36,7 +36,7 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
@@ -44,15 +44,15 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
@@ -60,8 +60,8 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
auto Src1 = SHADataShuffle(Dest);
@@ -70,7 +70,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
// The result is swizzled differently than expected
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
@@ -79,8 +79,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
return;
}
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result {};
Ref ConstantVector {};
@@ -112,7 +112,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
}
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
@@ -120,12 +120,12 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto Result = _VSha256U0(Dest, Src);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
@@ -133,8 +133,8 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
@@ -142,7 +142,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
auto Result = _VSha256U1(Src1, Src2);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
@@ -150,8 +150,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// Hardcoded to XMM0
auto XMM0 = LoadXMMRegister(0);
@@ -177,7 +177,7 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
auto B = _VSha256H2(EFGH, ABCD, Key);
auto Result = shuffle_abcd(A, B);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
@@ -185,9 +185,9 @@ void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESImc(Src);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
@@ -195,10 +195,10 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
@@ -208,11 +208,11 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESENC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
@@ -220,10 +220,10 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
@@ -233,11 +233,11 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESENCLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
@@ -245,10 +245,10 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
@@ -258,11 +258,11 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESDEC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
@@ -270,10 +270,10 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
@@ -283,15 +283,15 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESDECLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const uint64_t RCON = Op->Src[1].Literal();
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
@@ -305,7 +305,7 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
}
Ref Result = AESKeyGenAssistImpl(Op);
StoreResultFPR(Op, Result);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
@@ -313,12 +313,12 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
StoreResultFPR(Op, Res);
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
}
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
@@ -328,12 +328,12 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
}
const auto DstSize = OpSizeFromDst(Op);
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
StoreResultFPR(Op, Res);
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
}
} // namespace FEXCore::IR
@@ -263,7 +263,7 @@ void OpDispatchBuilder::CalculateDeferredFlags() {
Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
// If CF not inverted, we use .cc since the increment happens when the
// condition is false. If CF inverted, invert to use .cs. A bit mindbendy.
return _NZCVSelectIncrement(OpSize, CFInverted ? CondClass::UGE : CondClass::ULT, Src, Src);
return _NZCVSelectIncrement(OpSize, {CFInverted ? COND_UGE : COND_ULT}, Src, Src);
}
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
@@ -290,7 +290,7 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
// TODO: We can fold that second Bfe in (cmp uxth).
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Res, Src2PlusCF);
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Res, Src2PlusCF);
SetNZ_ZeroCV(SrcSize, Res);
SetCFInverted(SelectCFInv);
@@ -324,7 +324,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
Res = Sub(OpSize, Src1, Src2PlusCF);
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Src1, Src2PlusCF);
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Src1, Src2PlusCF);
SetNZ_ZeroCV(SrcSize, Res);
SetCFInverted(SelectCFInv);
@@ -406,7 +406,7 @@ void OpDispatchBuilder::CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High
// If High = SignBit, then sets to nZCv. Else sets to nzcV. Since SF/ZF
// undefined, this does what we need after inverting carry.
auto Zero = _InlineConstant(0);
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
CFInverted = true;
}
@@ -423,7 +423,7 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
// If High = 0, then sets to nZCv. Else sets to nzcV. Since SF/ZF undefined,
// this does what we need.
_CondSubNZCV(Size, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
_CondSubNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
CFInverted = true;
}
@@ -151,9 +151,6 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 4), 4, &OpDispatchBuilder::NOPOp},
// GROUP 17
{OPD(FEXCore::X86Tables::TYPE_GROUP_17, PF_66, 0), 1, &OpDispatchBuilder::Extrq_imm},
// GROUP P
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, true, false, 1>},
@@ -198,8 +198,6 @@ constexpr DispatchTableEntry OpDispatch_SecondaryRepNEModTables[] = {
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
{0x78, 1, &OpDispatchBuilder::Insertq_imm},
{0x79, 1, &OpDispatchBuilder::Insertq},
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i32Bit>},
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
@@ -258,7 +256,6 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>},
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
{0x78, 1, nullptr}, // GROUP 17
{0x79, 1, &OpDispatchBuilder::Extrq},
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>},
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i64Bit>},
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
File diff suppressed because it is too large. Load diff
@@ -28,12 +28,10 @@ class OrderedNode;
Ref OpDispatchBuilder::GetX87Top() {
// Yes, we are storing 3 bits in a single flag register.
// Deal with it
return _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
}
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
_StackForceSlow(); // Invalidate x87 FTW register cache
// For the output, we want a 1-bit for each pair not equal to 11 (Empty).
static_assert(static_cast<uint8_t>(FPState::X87Tag::Empty) == 0b11);
@@ -52,18 +50,18 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4);
// ...and that's it. StoreContext implicitly does the final masking.
_StoreContextGPR(OpSize::i8Bit, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, FTW);
}
void OpDispatchBuilder::SetX87Top(Ref Value) {
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
}
// Float LoaD operation with memory operand
void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) {
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
Ref ConvertedData = Data;
// Convert to 80bit float
if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
@@ -79,14 +77,14 @@ void OpDispatchBuilder::FLDFromStack(OpcodeArgs) {
void OpDispatchBuilder::FBLD(OpcodeArgs) {
// Read from memory
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
Ref ConvertedData = _F80BCDLoad(Data);
_PushStack(ConvertedData, Data, OpSize::i128Bit, true);
}
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
Ref converted = _F80BCDStore(_ReadStackValue(0));
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
_PopStackDestroy();
}
@@ -99,7 +97,7 @@ void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant K) {
void OpDispatchBuilder::FILD(OpcodeArgs) {
const auto ReadWidth = OpSizeFromSrc(Op);
// Read from memory
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
// Sign extend to 64bits
if (ReadWidth != OpSize::i64Bit) {
@@ -112,15 +110,15 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
// Extract sign and make integer absolute
auto zero = Constant(0);
_SubNZCV(OpSize::i64Bit, Data, zero);
auto sign = _NZCVSelect(OpSize::i64Bit, CondClass::SLT, Constant(0x8000), zero);
auto absolute = _Neg(OpSize::i64Bit, Data, CondClass::MI);
auto sign = _NZCVSelect(OpSize::i64Bit, CondClassType {COND_SLT}, Constant(0x8000), zero);
auto absolute = _Neg(OpSize::i64Bit, Data, CondClassType {COND_MI});
// left justify the absolute integer
auto shift = Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
auto shifted = _Lshl(OpSize::i64Bit, absolute, shift);
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
auto zeroed_exponent = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, absolute, zero, zero, adjusted_exponent);
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
Ref ConvertedData = _VLoadTwoGPRs(shifted, upper);
@@ -166,12 +164,12 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
// Check for NaN/Infinity: exponent = 0x7fff
SaveNZCV();
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff));
Ref IsSpecial = _NZCVSelect01(CondClass::EQ);
Ref IsSpecial = _NZCVSelect01({COND_EQ});
// For overflow detection, check if exponent indicates a value >= 2^15
// Biased exponent for 2^15 is 0x3fff + 15 = 0x400e
SubWithFlags(OpSize::i64Bit, Exponent, 0x400e);
Ref IsOverflow = _NZCVSelect01(CondClass::UGE);
Ref IsOverflow = _NZCVSelect01({COND_UGE});
// Set Invalid Operation flag if overflow or special value
Ref InvalidFlag = _Or(OpSize::i64Bit, IsSpecial, IsOverflow);
@@ -180,7 +178,7 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
Data = _F80CVTInt(Size, Data, Truncate);
StoreResultGPR_WithOpSize(Op, Op->Dest, Data, Size, OpSize::i8Bit);
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, OpSize::i8Bit);
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
_PopStackDestroy();
@@ -206,10 +204,10 @@ void OpDispatchBuilder::FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
// We have one memory argument
Ref Arg {};
if (Integer) {
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
Arg = _F80CVTToInt(Arg, Width);
} else {
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Arg = _F80CVTTo(Arg, Width);
}
@@ -236,10 +234,10 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
// We have one memory argument
Ref arg {};
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
arg = _F80CVTToInt(arg, Width);
} else {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _F80CVTTo(arg, Width);
}
@@ -273,10 +271,10 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
// We have one memory argument
Ref arg {};
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
arg = _F80CVTToInt(arg, Width);
} else {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _F80CVTTo(arg, Width);
}
@@ -314,10 +312,10 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
// We have one memory argument
Ref Arg {};
if (Integer) {
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
Arg = _F80CVTToInt(Arg, Width);
} else {
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Arg = _F80CVTTo(Arg, Width);
}
@@ -340,7 +338,7 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() {
// bytes, we use the well-known bit twiddling algorithm:
//
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
Ref X = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref X = LoadContext(AbridgedFTWIndex);
X = _Orlshl(OpSize::i32Bit, X, X, 4);
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f));
X = _Orlshl(OpSize::i32Bit, X, X, 2);
@@ -381,41 +379,41 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
_SyncStackToSlow();
const auto Size = OpSizeFromSrc(Op);
Ref Mem = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
Mem = AppendSegmentOffset(Mem, Op->Flags);
{
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMemGPR(Size, Mem, FCW, Size);
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMem(GPRClass, Size, Mem, FCW, Size);
}
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
auto ZeroConst = Constant(0);
{
// FTW
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
}
{
// Instruction Offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
}
{
// Instruction CS selector (+ Opcode)
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
}
{
// Data pointer offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
}
{
// Data pointer selector
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
}
}
@@ -441,20 +439,20 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
_StackForceSlow();
const auto Size = OpSizeFromSrc(Op);
Ref Mem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
Ref Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
Mem = AppendSegmentOffset(Mem, Op->Flags);
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 1);
auto NewFSW = _LoadMemGPR(Size, MemLocation, Size);
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
ReconstructX87StateFromFSW_Helper(NewFSW);
{
// FTW
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 2);
SetX87FTW(_LoadMemGPR(Size, MemLocation, Size));
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
}
}
@@ -483,61 +481,61 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
Ref Top = GetX87Top();
{
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMemGPR(Size, Mem, FCW, Size);
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMem(GPRClass, Size, Mem, FCW, Size);
}
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
auto ZeroConst = Constant(0);
{
// FTW
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
}
{
// Instruction Offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
}
{
// Instruction CS selector (+ Opcode)
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
}
{
// Data pointer offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
}
{
// Data pointer selector
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
}
auto SevenConst = Constant(7);
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (int i = 0; i < 7; ++i) {
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
if (ReducedPrecisionMode) {
data = _F80CVTTo(data, OpSize::i64Bit);
}
_StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreMem(FPRClass, OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
}
// The final st(7) needs a bit of special handling here
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
if (ReducedPrecisionMode) {
data = _F80CVTTo(data, OpSize::i64Bit);
}
// ST7 broken in to two parts
// Lower 64bits [63:0]
// upper 16 bits [79:64]
_StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreMem(FPRClass, OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4);
_StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreMem(FPRClass, OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
// reset to default
FNINIT(Op);
@@ -548,8 +546,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
const auto Size = OpSizeFromSrc(Op);
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
if (ReducedPrecisionMode) {
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
@@ -561,11 +559,11 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
_SetRoundingMode(roundingMode, false, roundingMode);
}
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1);
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1);
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
{
// FTW
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
}
auto SevenConst = Constant(7);
@@ -574,14 +572,14 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
Ref Mask = _VLoadTwoGPRs(low, high);
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (int i = 0; i < 7; ++i) {
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
// Mask off the top bits
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
if (ReducedPrecisionMode) {
// Convert to double precision
Reg = _F80CVT(OpSize::i64Bit, Reg);
}
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
}
@@ -590,19 +588,20 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
// ST7 broken in to two parts
// Lower 64bits [63:0]
// upper 16 bits [79:64]
Ref Reg = _LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Ref RegHigh =
_LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
if (ReducedPrecisionMode) {
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
}
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
}
// Load / Store Control Word
void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
StoreResultGPR(Op, FCW);
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
StoreResult(GPRClass, Op, FCW, OpSize::iInvalid);
}
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
@@ -610,8 +609,8 @@ void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
// to switch for now to slow mode whenever these are manually changed.
// Remove the next line and try DF_04.asm in fast path.
_StackForceSlow();
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
}
void OpDispatchBuilder::FXCH(OpcodeArgs) {
@@ -647,10 +646,10 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
// Memory arg
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
b = _F80CVTToInt(arg, Width);
} else {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
b = _F80CVTTo(arg, Width);
}
} else {
@@ -766,7 +765,7 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
Ref TopValue = _SyncStackToSlow();
Ref StatusWord = ReconstructFSW_Helper(TopValue);
StoreResultGPR(Op, StatusWord);
StoreResult(GPRClass, Op, StatusWord, OpSize::iInvalid);
}
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
@@ -775,8 +774,6 @@ void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
}
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
_SyncStackToSlow(); // Invalidate x87 register caches
auto Zero = Constant(0);
if (ReducedPrecisionMode) {
@@ -785,12 +782,12 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
// Init FCW to 0x037F
auto NewFCW = Constant(0x037F);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
// Set top to zero
SetX87Top(Zero);
// Tags all get marked as invalid
_StoreContextGPR(OpSize::i8Bit, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, Zero);
// Reinits the simulated stack
_InitStack();
@@ -863,7 +860,7 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
auto TopValid = _StackValidTag(0);
// In the case of top being invalid then C3:C2:C0 is 0b101
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
auto C3 = Select01(OpSize::i32Bit, CondClassType {COND_NEQ}, TopValid, Constant(1));
auto C2 = TopValid;
auto C0 = C3; // Mirror C3 until something other than zero is supported
@@ -29,38 +29,38 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
const auto Size = OpSizeFromSrc(Op);
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
_SetRoundingMode(roundingMode, false, roundingMode);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1);
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MEM_OFFSET_SXTX, 1);
ReconstructX87StateFromFSW_Helper(NewFSW);
{
// FTW
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
}
}
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
_StackForceSlow();
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
_SetRoundingMode(roundingMode, false, roundingMode);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
}
// F64 ops
// Float load op with memory operand
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
// Convert to 64bit float
Ref ConvertedData = Data;
if (Width == OpSize::i32Bit) {
@@ -73,7 +73,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
// Read from memory
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i128Bit, Op->Flags);
Ref ConvertedData = _F80BCDLoad(Data);
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
@@ -82,7 +82,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
Ref converted = _F80CVTTo(_ReadStackValue(0), OpSize::i64Bit);
converted = _F80BCDStore(converted);
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
_PopStackDestroy();
}
@@ -95,7 +95,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
const auto ReadWidth = OpSizeFromSrc(Op);
// Read from memory
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
if (ReadWidth == OpSize::i16Bit) {
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
}
@@ -112,7 +112,7 @@ void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
} else {
data = _Float_ToGPR_S(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data);
}
StoreResultGPR_WithOpSize(Op, Op->Dest, data, Size, OpSize::i8Bit);
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, OpSize::i8Bit);
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
_PopStackDestroy();
@@ -138,16 +138,16 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
Ref arg {};
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
} else if (Width == OpSize::i32Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
} else if (Width == OpSize::i64Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
} else {
FEX_UNREACHABLE;
}
@@ -176,16 +176,16 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
Ref arg {};
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
} else if (Width == OpSize::i32Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
} else if (Width == OpSize::i64Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
} else {
FEX_UNREACHABLE;
}
@@ -228,16 +228,16 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
if (Integer) {
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
Arg = _Sbfe(OpSize::i64Bit, 16, 0, Arg);
}
Arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Arg);
} else if (Width == OpSize::i32Bit) {
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Arg);
} else if (Width == OpSize::i64Bit) {
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
}
} else {
FEX_UNREACHABLE;
@@ -285,16 +285,16 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
} else if (Width == OpSize::i32Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
} else if (Width == OpSize::i64Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
}
} else {
FEX_UNREACHABLE;
@@ -332,16 +332,16 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
} else if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
// Memory arg
if (Integer) {
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (Width == OpSize::i16Bit) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
} else if (Width == OpSize::i32Bit) {
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
} else if (Width == OpSize::i64Bit) {
b = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
}
} else {
FEX_UNREACHABLE;
@@ -393,8 +393,8 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
SaveNZCV();
_TestNZ(OpSize::i64Bit, Gpr, Constant(0x7fff'ffff'ffff'ffffUL));
Ref Sig = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, SigZV, SigNZV);
Ref Exp = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, ExpZV, ExpNZV);
Ref Sig = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, SigZV, SigNZV);
Ref Exp = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, ExpZV, ExpNZV);
_PopStackDestroy();
_PushStack(Exp, Exp, OpSize::i64Bit, true);
@@ -442,7 +442,7 @@ constexpr std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() c
{0x70, 1, X86InstInfo{"PSHUFLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
{0x71, 3, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0}},
{0x74, 4, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{0x78, 1, X86InstInfo{"INSERTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS,2}},
{0x78, 1, X86InstInfo{"INSERTQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS,2}},
{0x79, 1, X86InstInfo{"INSERTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS, 0}},
{0x7A, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{0x7C, 1, X86InstInfo{"HADDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
@@ -7,7 +7,6 @@ $end_info$
#pragma once
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
@@ -43,9 +42,8 @@ constexpr uint32_t FLAG_DS_PREFIX = (0b100 << 11);
constexpr uint32_t FLAG_FS_PREFIX = (0b101 << 11);
constexpr uint32_t FLAG_GS_PREFIX = (0b110 << 11);
constexpr uint32_t FLAG_SEGMENTS = (0b111 << 11);
constexpr uint32_t FLAG_FORCE_TSO = (1 << 14);
constexpr uint32_t FLAG_DECODED_MODRM = (1 << 15);
constexpr uint32_t FLAG_DECODED_SIB = (1 << 16);
// Bits 14, 15, 16 - Unused
constexpr uint32_t FLAG_REP_PREFIX = (1 << 17);
constexpr uint32_t FLAG_REPNE_PREFIX = (1 << 18);
// Size flags
@@ -145,9 +143,6 @@ struct DecodedOperand {
}
uint64_t Literal() const {
LOGMAN_THROW_A_FMT(IsLiteral(), "Precondition: must be a literal");
if (Data.Literal.SignExtend) {
return static_cast<int64_t>(static_cast<int32_t>(Data.Literal.Value));
}
return Data.Literal.Value;
}
@@ -171,9 +166,8 @@ struct DecodedOperand {
} RIPLiteral;
struct LiteralType {
uint32_t Value;
uint8_t Size : 7 ;
bool SignExtend : 1;
uint64_t Value;
uint8_t Size;
auto operator<=>(const LiteralType&) const = default;
} Literal;
@@ -199,12 +193,16 @@ struct DecodedInst {
X86InstInfo const* TableInfo;
uint32_t Flags;
uint16_t OP;
uint8_t OPRaw;
uint16_t OP;
uint8_t ModRM;
uint8_t SIB;
uint8_t InstSize;
uint8_t LastEscapePrefix;
bool DecodedModRM;
bool DecodedSIB;
bool ForceTSO;
};
union ModRMDecoded {
@@ -560,6 +558,21 @@ constexpr static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86T
}
};
template<typename OpcodeType>
static inline void LateInitCopyTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *OtherLocal, size_t OtherTableSize) {
for (size_t j = 0; j < OtherTableSize; ++j) {
X86TablesInfoStruct<OpcodeType> const &OtherOp = OtherLocal[j];
auto OtherOpNum = OtherOp.first;
X86InstInfo const &OtherInfo = OtherOp.Info;
for (uint32_t i = 0; i < OtherOp.second; ++i) {
X86InstInfo &FinalOp = FinalTable[OtherOpNum + i];
if (FinalOp.Type == TYPE_COPY_OTHER) {
FinalOp = OtherInfo;
}
}
}
}
template<typename OpcodeType>
constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
for (size_t j = 0; j < TableSize; ++j) {
@@ -590,6 +603,12 @@ constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86Tables
}
};
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::X86Tables::DecodedOperand::OpType);
}
} // namespace FEXCore::X86Tables
template <>
struct fmt::formatter<FEXCore::X86Tables::DecodedOperand::OpType> : formatter<uint32_t> {
template <typename FormatContext>
auto format(FEXCore::X86Tables::DecodedOperand::OpType type, FormatContext& ctx) const {
return fmt::formatter<uint32_t>::format(static_cast<uint32_t>(type), ctx);
}
};
+7 -7
View File
@@ -42,19 +42,19 @@ void __attribute__((noinline)) __jit_debug_register_code() {
namespace FEXCore {
void GDBJITRegister(FEXCore::ExecutableFileInfo& Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
FEXCore::Core::DebugData& DebugData) {
auto map = Entry.SourcecodeMap.get();
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
FEXCore::Core::DebugData* DebugData) {
auto map = Entry->SourcecodeMap.get();
if (map) {
auto FileOffset = GuestRIP - VAFileStart;
auto Sym = map->FindSymbolMapping(FileOffset);
auto SymName = HLE::SourcecodeSymbolMapping::SymName(Sym, Entry.Filename, HostEntry, FileOffset);
auto SymName = HLE::SourcecodeSymbolMapping::SymName(Sym, Entry->Filename, HostEntry, FileOffset);
fextl::vector<gdb_line_mapping> Lines;
for (const auto& GuestOpcode : DebugData.GuestOpcodes) {
for (const auto& GuestOpcode : DebugData->GuestOpcodes) {
auto Line = map->FindLineMapping(GuestRIP + GuestOpcode.GuestEntryOffset - VAFileStart);
if (Line) {
Lines.push_back({Line->LineNumber, HostEntry + GuestOpcode.HostEntryOffset});
@@ -80,7 +80,7 @@ void GDBJITRegister(FEXCore::ExecutableFileInfo& Entry, uintptr_t VAFileStart, u
for (int i = 0; i < info->nblocks; i++) {
strncpy(blocks[i].name, SymName.c_str(), 511);
blocks[i].start = HostEntry;
blocks[i].end = HostEntry + DebugData.HostCodeSize;
blocks[i].end = HostEntry + DebugData->HostCodeSize;
}
info->nlines = Lines.size();
@@ -113,7 +113,7 @@ void GDBJITRegister(FEXCore::ExecutableFileInfo& Entry, uintptr_t VAFileStart, u
} // namespace FEXCore
#else
namespace FEXCore {
void GDBJITRegister(FEXCore::ExecutableFileInfo&, uintptr_t, uint64_t, uintptr_t, FEXCore::Core::DebugData&) {
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry*, uintptr_t, uint64_t, uintptr_t, FEXCore::Core::DebugData*) {
ERROR_AND_DIE_FMT("GDBSymbols support not compiled in");
}
} // namespace FEXCore
+5 -4
View File
@@ -1,8 +1,9 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/Core/CodeCache.h>
#include <Interface/Core/JIT/DebugData.h>
#include <Interface/IR/AOTIR.h>
namespace FEXCore {
void GDBJITRegister(FEXCore::ExecutableFileInfo&, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry, FEXCore::Core::DebugData&);
}
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
FEXCore::Core::DebugData* DebugData);
}
+60
View File
@@ -0,0 +1,60 @@
// SPDX-License-Identifier: MIT
#include "FEXHeaderUtils/Filesystem.h"
#include "Interface/Context/Context.h"
#include "Interface/IR/AOTIR.h"
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/fextl/fmt.h>
#include <Interface/Core/LookupCache.h>
#include <Interface/GDBJIT/GDBJIT.h>
#include <xxhash.h>
namespace FEXCore::IR {
bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr,
uint64_t Length, FEXCore::Core::DebugData* DebugData) {
// Both generated ir and LibraryJITName need a named region lookup
if (CTX->Config.LibraryJITNaming() || CTX->Config.GDBSymbols()) {
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (AOTIRCacheEntry.Entry) {
if (DebugData && CTX->Config.LibraryJITNaming()) {
CTX->Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, AOTIRCacheEntry.Entry->Filename);
}
if (CTX->Config.GDBSymbols()) {
GDBJITRegister(AOTIRCacheEntry.Entry, AOTIRCacheEntry.VAFileStart, GuestRIP, (uintptr_t)CodePtr, DebugData);
}
}
}
return false;
}
AOTIRCacheEntry* AOTIRCaptureCache::LoadAOTIRCacheEntry(const fextl::string& filename) {
fextl::string base_filename = FHU::Filesystem::GetFilename(filename);
if (!base_filename.empty()) {
auto filename_hash = XXH3_64bits(filename.c_str(), filename.size());
auto fileid = fextl::fmt::format("{}-{}-{}{}{}", base_filename, filename_hash,
(CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) ? 'S' : 's',
CTX->Config.TSOEnabled ? 'T' : 't', CTX->Config.ABILocalFlags ? 'L' : 'l');
std::unique_lock lk(AOTIRCacheLock);
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry {.FileId = fileid, .Filename = filename}});
auto Entry = &(Inserted.first->second);
return Entry;
}
return nullptr;
}
} // namespace FEXCore::IR
+72
View File
@@ -0,0 +1,72 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/HLE/SourcecodeResolver.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <cstdint>
#include <shared_mutex>
namespace FEXCore::CPU {
union Relocation;
} // namespace FEXCore::CPU
namespace FEXCore::Core {
struct InternalThreadState;
struct DebugDataSubblock {
uint32_t HostCodeOffset;
uint32_t HostCodeSize;
};
struct DebugDataGuestOpcode {
uint64_t GuestEntryOffset;
ptrdiff_t HostEntryOffset;
};
/**
* @brief Contains debug data for a block of code for later debugger analysis
*
* Needs to remain around for as long as the code could be executed at least
*/
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
fextl::vector<DebugDataSubblock> Subblocks;
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
};
} // namespace FEXCore::Core
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::IR {
struct AOTIRCacheEntry {
fextl::unique_ptr<FEXCore::HLE::SourcecodeMap> SourcecodeMap;
fextl::string FileId;
fextl::string Filename;
};
class AOTIRCaptureCache final {
public:
AOTIRCaptureCache(FEXCore::Context::ContextImpl* ctx)
: CTX {ctx} {}
bool PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr, uint64_t Length,
FEXCore::Core::DebugData* DebugData);
AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& filename);
private:
FEXCore::Context::ContextImpl* CTX;
std::shared_mutex AOTIRCacheLock;
fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCacheEntry> AOTIRCache;
};
} // namespace FEXCore::IR
+135 -18
View File
@@ -1,9 +1,7 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/ThreadPoolAllocator.h>
#include <FEXCore/IR/IR.h>
@@ -11,15 +9,11 @@
#include <FEXCore/fextl/sstream.h>
#include <array>
#include <cstddef>
#include <cstdint>
#include <functional>
#include <iterator>
#include <type_traits>
namespace FEXCore::IR {
class OrderedNode;
class RegisterAllocationPass;
/**
* @brief The IROp_Header is an dynamically sized array
@@ -241,6 +235,8 @@ static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3);
* The second region is contiguous but they don't have any relationship with one another directly
*/
class OrderedNode final {
friend class NodeWrapperIterator;
friend class OrderedList;
public:
// These three values are laid out very specifically to make it fast to access the NodeWrappers specifically
OrderedNodeHeader Header;
@@ -430,6 +426,94 @@ static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + 2 * sizeof(uin
// };
using Ref = OrderedNode*;
struct FEX_PACKED RegisterClassType final {
using value_type = uint32_t;
value_type Val;
[[nodiscard]] constexpr operator value_type() const {
return Val;
}
[[nodiscard]]
friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
};
struct FEX_PACKED CondClassType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]]
friend constexpr bool operator==(const CondClassType&, const CondClassType&) = default;
};
struct FEX_PACKED MemOffsetType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]]
friend constexpr bool operator==(const MemOffsetType&, const MemOffsetType&) = default;
};
struct FEX_PACKED TypeDefinition final {
uint16_t Val;
[[nodiscard]] constexpr operator uint16_t() const {
return Val;
}
[[nodiscard]]
static constexpr TypeDefinition Create(uint8_t Bytes) {
TypeDefinition Type {};
Type.Val = Bytes << 8;
return Type;
}
[[nodiscard]]
static constexpr TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
TypeDefinition Type {};
Type.Val = (Bytes << 8) | (Elements & 255);
return Type;
}
[[nodiscard]]
constexpr uint8_t Bytes() const {
return Val >> 8;
}
[[nodiscard]]
constexpr uint8_t Elements() const {
return Val & 255;
}
[[nodiscard]]
friend constexpr bool operator==(const TypeDefinition&, const TypeDefinition&) = default;
};
static_assert(std::is_trivially_copyable_v<TypeDefinition>);
struct FEX_PACKED FenceType final {
using value_type = uint8_t;
value_type Val;
[[nodiscard]] constexpr operator value_type() const {
return Val;
}
[[nodiscard]]
friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
};
struct FEX_PACKED RoundType final {
uint8_t Val;
[[nodiscard]] constexpr operator uint8_t() const {
return Val;
}
[[nodiscard]]
friend constexpr bool operator==(const RoundType&, const RoundType&) = default;
};
class NodeIterator;
/* This iterator can be used to step though nodes.
* Due to how our IR is laid out, this can be used to either step
* though the CodeBlocks or though the code within a single block.
@@ -437,8 +521,8 @@ using Ref = OrderedNode*;
class NodeIterator {
public:
struct value_type final {
OrderedNode* Node;
IROp_Header* Header;
OrderedNode *Node;
IROp_Header *Header;
};
using size_type = std::size_t;
using difference_type = std::ptrdiff_t;
@@ -693,15 +777,6 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
bool IsBlockExit(FEXCore::IR::IROps Op);
void Dump(fextl::stringstream* out, const IRListView* IR);
constexpr auto format_as(FEXCore::IR::NodeID ID) {
return ID.Value;
}
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::FenceType)
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::MemOffsetType)
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::OpSize)
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::RegClass)
} // namespace FEXCore::IR
template<>
@@ -710,3 +785,45 @@ struct std::hash<FEXCore::IR::NodeID> {
return std::hash<FEXCore::IR::NodeID::value_type> {}(ID.Value);
}
};
template<>
struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID::value_type> {
using Base = fmt::formatter<FEXCore::IR::NodeID::value_type>;
// Pass-through the underlying value, so IDs can
// be formatted like any integral value.
template<typename FormatContext>
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
return Base::format(ID.Value, ctx);
}
};
template<>
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
template<typename FormatContext>
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
return Base::format(Class.Val, ctx);
}
};
template<>
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
template<typename FormatContext>
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
return Base::format(Fence.Val, ctx);
}
};
template<>
struct fmt::formatter<FEXCore::IR::OpSize> : fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>> {
using Base = fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>>;
template<typename FormatContext>
auto format(const FEXCore::IR::OpSize& OpSize, FormatContext& ctx) const {
return Base::format(FEXCore::ToUnderlying(OpSize), ctx);
}
};
+109 -123
View File
@@ -52,68 +52,80 @@
" * These are validations that can't be automatically inferred and need to be hand-written",
""
],
"Enums": {
"class CondClass : uint8_t": [
"EQ = 0,",
"NEQ = 1,",
"UGE = 2,",
"ULT = 3,",
"MI = 4,",
"PL = 5,",
"VS = 6,",
"VC = 7,",
"UGT = 8,",
"ULE = 9,",
"SGE = 10,",
"SLT = 11,",
"SGT = 12,",
"SLE = 13,",
"TSTZ = 14, /* bit test zero */",
"TSTNZ = 15, /* bit test nonzero */",
"",
"FLU = 16, /* float less or unordered */",
"FGE = 17, /* float greater or equal */",
"FLEU = 18, /* float less or equal or unordered */",
"FGT = 19, /* float greater */",
"FU = 20, /* float unordered */",
"FNU = 21, /* float not unordered */",
"",
"AL = 32, /* always */"
],
"class FenceType : uint8_t": [
"Load = 0,",
"Store = 1,",
"LoadStore = 2,",
"Inst = 3,"
],
"class MemOffsetType : uint8_t": [
"SXTX = 0,",
"UXTW = 1,",
"SXTW = 2,"
],
"class RegClass : uint32_t": [
"Invalid = 0,",
"GPR = 1,",
"GPRFixed = 2,",
"FPR = 3,",
"FPRFixed = 4,",
"Complex = 5,"
],
"class RoundMode : uint8_t": [
"Nearest = 0,",
"NegInfinity = 1,",
"PosInfinity = 2,",
"TowardsZero = 3, /* Truncate */",
"Host = 4,"
]
},
"Defines": [
"constexpr uint8_t NumClasses {6}",
"constexpr uint8_t COND_EQ = 0",
"constexpr uint8_t COND_NEQ = 1",
"constexpr uint8_t COND_UGE = 2",
"constexpr uint8_t COND_ULT = 3",
"constexpr uint8_t COND_MI = 4",
"constexpr uint8_t COND_PL = 5",
"constexpr uint8_t COND_VS = 6",
"constexpr uint8_t COND_VC = 7",
"constexpr uint8_t COND_UGT = 8",
"constexpr uint8_t COND_ULE = 9",
"constexpr uint8_t COND_SGE = 10",
"constexpr uint8_t COND_SLT = 11",
"constexpr uint8_t COND_SGT = 12",
"constexpr uint8_t COND_SLE = 13",
"constexpr uint8_t COND_TSTZ = 14 /* bit test zero */",
"constexpr uint8_t COND_TSTNZ = 15 /* bit test nonzero */",
"constexpr uint8_t COND_FLU = 16 /* float less or unordred */",
"constexpr uint8_t COND_FGE = 17 /* float greater or equal */",
"constexpr uint8_t COND_FLEU = 18 /* float less or equal or unordred */",
"constexpr uint8_t COND_FGT = 19 /* float greater */",
"constexpr uint8_t COND_FU = 20 /* float unordred */",
"constexpr uint8_t COND_FNU = 21 /* float not unordred */",
"constexpr uint8_t COND_AL = 32 /* always */",
"constexpr FEXCore::IR::RegisterClassType InvalidClass {0}",
"constexpr FEXCore::IR::RegisterClassType GPRClass {1}",
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {2}",
"constexpr FEXCore::IR::RegisterClassType FPRClass {3}",
"constexpr FEXCore::IR::RegisterClassType FPRFixedClass {4}",
"constexpr FEXCore::IR::RegisterClassType ComplexClass {5}",
"constexpr uint8_t NumClasses {6}",
"",
"constexpr FEXCore::IR::TypeDefinition i8 {TypeDefinition::Create(1, 0)}",
"constexpr FEXCore::IR::TypeDefinition i16 {TypeDefinition::Create(2, 0)}",
"constexpr FEXCore::IR::TypeDefinition i32 {TypeDefinition::Create(4, 0)}",
"constexpr FEXCore::IR::TypeDefinition i64 {TypeDefinition::Create(8, 0)}",
"constexpr FEXCore::IR::TypeDefinition i128 {TypeDefinition::Create(16, 0)}",
"",
"constexpr FEXCore::IR::TypeDefinition i8v8 {TypeDefinition::Create(1, 8)}",
"constexpr FEXCore::IR::TypeDefinition i8v16 {TypeDefinition::Create(1, 16)}",
"constexpr FEXCore::IR::TypeDefinition i16v4 {TypeDefinition::Create(2, 4)}",
"constexpr FEXCore::IR::TypeDefinition i16v8 {TypeDefinition::Create(2, 8)}",
"constexpr FEXCore::IR::TypeDefinition i32v2 {TypeDefinition::Create(4, 2)}",
"constexpr FEXCore::IR::TypeDefinition i32v4 {TypeDefinition::Create(4, 4)}",
"constexpr FEXCore::IR::TypeDefinition i64v2 {TypeDefinition::Create(8, 2)}",
"",
"constexpr uint8_t FCMP_FLAG_EQ = 0",
"constexpr uint8_t FCMP_FLAG_LT = 1",
"constexpr uint8_t FCMP_FLAG_UNORDERED = 2",
"constexpr FEXCore::IR::FenceType Fence_Load {0}",
"constexpr FEXCore::IR::FenceType Fence_Store {1}",
"constexpr FEXCore::IR::FenceType Fence_LoadStore {2}",
"constexpr FEXCore::IR::FenceType Fence_Inst {3}",
"constexpr uint8_t ROUND_MODE_NEAREST = 0",
"constexpr uint8_t ROUND_MODE_NEGATIVE_INFINITY = 1",
"constexpr uint8_t ROUND_MODE_POSITIVE_INFINITY = 2",
"constexpr uint8_t ROUND_MODE_TOWARDS_ZERO = 3",
"constexpr uint8_t ROUND_MODE_FLUSH_TO_ZERO = 1 << 2",
"constexpr FEXCore::IR::RoundType Round_Nearest {ROUND_MODE_NEAREST}",
"constexpr FEXCore::IR::RoundType Round_Negative_Infinity {ROUND_MODE_NEGATIVE_INFINITY}",
"constexpr FEXCore::IR::RoundType Round_Positive_Infinity {ROUND_MODE_POSITIVE_INFINITY}",
"constexpr FEXCore::IR::RoundType Round_Towards_Zero {ROUND_MODE_TOWARDS_ZERO} /* Truncate */",
"constexpr FEXCore::IR::RoundType Round_Host {ROUND_MODE_TOWARDS_ZERO + 1}",
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_SXTX {0}",
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_UXTW {1}",
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_SXTW {2}",
"struct BreakDefinition {",
" uint16_t ErrorRegister;",
" uint8_t Signal;",
@@ -136,13 +148,13 @@
"GPR": "OrderedNode*",
"FPR": "OrderedNode*",
"FenceType": "FenceType",
"RegisterClass": "RegClass",
"CondClass": "CondClass",
"RegisterClass": "RegisterClassType",
"CondClass": "CondClassType",
"SyscallFlags": "FEXCore::IR::SyscallFlags",
"SHA256Sum": "SHA256Sum",
"MemOffsetType": "MemOffsetType",
"BreakDefinition": "BreakDefinition",
"RoundType": "RoundMode",
"RoundType": "RoundType",
"FloatCompareOp": "FloatCompareOp",
"NamedVectorConstant": "FEXCore::IR::NamedVectorConstant",
"IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant",
@@ -295,7 +307,7 @@
"HasSideEffects": true,
"RAOverride": "0"
},
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{CondClass::NEQ}, OpSize:$CompareSize{OpSize::iInvalid}, i1:$FromNZCV{false}": {
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, OpSize:$CompareSize{OpSize::iInvalid}, i1:$FromNZCV{false}": {
"Inline": ["", "AddSub"],
"HasSideEffects": true,
"RAOverride": "2"
@@ -352,6 +364,20 @@
"GPR = Copy GPR:$Source": {
"Desc": ["GPR copy, generated by RA to split live ranges"],
"DestSize": "OpSize::i64Bit"
},
"GPR = Swap1 GPR:$A, GPR:$B": {
"Desc": ["GPR swap part 1, generated by RA. Returns value of first source.",
"Destination must be second GPR."],
"DestSize": "OpSize::i64Bit"
},
"GPR = Swap2": {
"Desc": ["GPR swap part 2, generated by RA. Returns source source.",
"Must immediately succeed Swap1 with no intervening instructions",
"Kludge to workaround single destination restriction on IR",
"Hopefully temporary"],
"DestSize": "OpSize::i64Bit"
}
},
"StaticRA": {
@@ -397,8 +423,8 @@
],
"DestSize": "ByteSize",
"EmitValidation": [
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
]
@@ -411,8 +437,8 @@
"HasSideEffects": true,
"DestSize": "ByteSize",
"EmitValidation": [
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
]
@@ -428,8 +454,8 @@
"HasSideEffects": true,
"DestSize": "ByteSize",
"EmitValidation": [
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
]
@@ -446,8 +472,8 @@
"EmitValidation": [
"WalkFindRegClass($Value1) == $Class",
"WalkFindRegClass($Value2) == $Class",
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
]
@@ -459,8 +485,8 @@
],
"DestSize": "ByteSize",
"EmitValidation": [
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContextIndexed to GPR\"",
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContextIndexed to XMM\""
]
@@ -473,23 +499,12 @@
"DestSize": "ByteSize",
"EmitValidation": [
"WalkFindRegClass($Value) == $Class",
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContextIndexed to GPR\"",
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContextIndexed to XMM\""
]
},
"GPR = FormContextAddress OpSize:#Size, GPR:$Index, u32:$Stride": {
"Desc": ["Forms an address into the context structure indexed by SSA value",
"Dest = Ctx + Index * Stride",
"This allows backends to compute the address once and reuse it for multiple memory operations",
"Stride must be a power of 2"
],
"DestSize": "Size",
"EmitValidation": [
"#Size == IR::OpSize::i64Bit"
]
},
"SpillRegister SSA:$Value, u32:$Slot, RegisterClass:$Class": {
"HasSideEffects": true,
@@ -606,7 +621,7 @@
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = VLoadVectorGatherMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$Mask, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, OpSize:$VectorIndexElementSize, u8:$OffsetScale, u8:$DataElementOffsetStart, u8:$IndexElementOffsetStart, OpSize:$AddrSize": {
"FPR = VLoadVectorGatherMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$Mask, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, OpSize:$VectorIndexElementSize, u8:$OffsetScale, u8:$DataElementOffsetStart, u8:$IndexElementOffsetStart": {
"Desc": [
"Does a masked load similar to VPGATHERD* where the upper bit of each element",
"determines whether or not that element will be loaded from memory.",
@@ -620,7 +635,7 @@
"$VectorIndexElementSize == OpSize::i32Bit || $VectorIndexElementSize == OpSize::i64Bit"
]
},
"FPR = VLoadVectorGatherMaskedQPS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$MaskReg, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, u8:$OffsetScale, OpSize:$AddrSize": {
"FPR = VLoadVectorGatherMaskedQPS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$MaskReg, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, u8:$OffsetScale": {
"Desc": [
"Does a masked load similar to VPGATHERQPS where the upper bit of each element",
"determines whether or not that element will be loaded from memory.",
@@ -735,10 +750,9 @@
},
"Fence FenceType:$Fence": {
"Desc": ["Does a memory fence operation of the desired type",
"FenceType::Load: Ensures load memory operations are serialized",
"FenceType::Store: Ensures store memory operations are serialized",
"FenceType::LoadStore: Ensures loads and store memory operations are serialized",
"FenceType::Inst: Instruction barrier. Ensures all instructions after this point will be explicitly fetched",
"Fence_Load: Ensures load memory operations are serialized",
"Fence_Store: Ensures store memory operations are serialized",
"Fence_LoadStore: Ensures loads and store memory operations are serialized",
"Ensures the memory operations are globally visible"
],
"HasSideEffects": true
@@ -983,7 +997,7 @@
"DestSize": "OpSize::i64Bit"
},
"GPR = Neg OpSize:#Size, GPR:$Src, CondClass:$Cond{CondClass::AL}": {
"GPR = Neg OpSize:#Size, GPR:$Src, CondClass:$Cond{{COND_AL}}": {
"Desc": ["Integer negation, with optional predication",
"Dest = Cond ? -Src : Src",
"Will truncate to 64 or 32bits"
@@ -1061,13 +1075,6 @@
"Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Rbit OpSize:#Size, GPR:$Src": {
"Desc": ["Reverses the bit order of the register"],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Add OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": [ "Integer Add",
"Will truncate to 64 or 32bits"
@@ -1535,15 +1542,6 @@
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = MaskGenerateFromBitWidth GPR:$BitWidth": {
"Desc": ["Generates a bit mask from with a value from [0, 63]",
"0 is special cased to full-mask",
"Special operation for SSE4a bitmask generation."
],
"DestSize": "FEXCore::IR::OpSize::i64Bit",
"ImplicitFlagClobber": true
},
"GPR = Extr OpSize:#Size, GPR:$Upper, GPR:$Lower, u8:$LSB": {
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
"It then extracts a bitfield width that size of a GPR from the LSB",
@@ -2152,34 +2150,22 @@
"FPR = VAnd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
"ElementSize": "ElementSize"
},
"FPR = VAndn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
"ElementSize": "ElementSize"
},
"FPR = VOr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
"ElementSize": "ElementSize"
},
"FPR = VXor OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
"ElementSize": "ElementSize"
},
"FPR = VUQAdd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
@@ -2843,7 +2829,7 @@
"Int: 64-bit, 32-bit, 16-bit"
],
"EmitValidation": [
"WalkFindRegClass($OriginalValue) == RegClass::FPR || WalkFindRegClass($OriginalValue) == RegClass::GPR"
"WalkFindRegClass($OriginalValue) == FPRClass || WalkFindRegClass($OriginalValue) == GPRClass"
],
"HasSideEffects": true,
"X87": true
+104 -149
View File
@@ -38,8 +38,8 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, uint64_t Arg)
*out << fextl::fmt::format("#{:#x}", Arg);
}
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg) {
if (Arg == CondClass::AL) {
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClassType Arg) {
if (Arg == COND_AL) {
*out << "ALWAYS";
return;
}
@@ -48,7 +48,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg)
"UGT", "ULE", "SGE", "SLT", "SGT", "SLE", "TSTZ", "TSTNZ",
"FLU", "FGE", "FLEU", "FGT", "FU", "FNU"};
*out << CondNames[FEXCore::ToUnderlying(Arg)];
*out << CondNames[Arg];
}
static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType Arg) {
@@ -58,39 +58,39 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType
"SXTW",
};
*out << Names[FEXCore::ToUnderlying(Arg)];
*out << Names[Arg];
}
static void PrintArg(fextl::stringstream* out, const IRListView*, RegClass Arg) {
*out << [Arg] {
switch (Arg) {
case RegClass::Invalid: return "Invalid";
case RegClass::GPR: return "GPR";
case RegClass::GPRFixed: return "GPRFixed";
case RegClass::FPR: return "FPR";
case RegClass::FPRFixed: return "FPRFixed";
case RegClass::Complex: return "Complex";
}
return "<Unknown RegClass Type>";
}();
static void PrintArg(fextl::stringstream* out, const IRListView*, RegisterClassType Arg) {
if (Arg == GPRClass.Val) {
*out << "GPR";
} else if (Arg == GPRFixedClass.Val) {
*out << "GPRFixed";
} else if (Arg == FPRClass.Val) {
*out << "FPR";
} else if (Arg == FPRFixedClass.Val) {
*out << "FPRFixed";
} else {
*out << "Unknown Registerclass " << Arg;
}
}
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
if (Arg.IsImmediate()) {
auto PhyReg = PhysicalRegister(Arg);
switch (PhyReg.AsRegClass()) {
case RegClass::GPR: *out << "r"; break;
case RegClass::GPRFixed: *out << "R"; break;
case RegClass::FPR: *out << "v"; break;
case RegClass::FPRFixed: *out << "V"; break;
case RegClass::Complex: *out << "c"; break;
case RegClass::Invalid: *out << "invalid"; break;
switch (PhyReg.Class) {
case FEXCore::IR::GPRClass.Val: *out << "r"; break;
case FEXCore::IR::GPRFixedClass.Val: *out << "R"; break;
case FEXCore::IR::FPRClass.Val: *out << "v"; break;
case FEXCore::IR::FPRFixedClass.Val: *out << "V"; break;
case FEXCore::IR::ComplexClass.Val: *out << "c"; break;
case FEXCore::IR::InvalidClass.Val: *out << "invalid"; break;
default: *out << "unknown"; break;
}
if (PhyReg.AsRegClass() != RegClass::Invalid) {
*out << std::dec << uint32_t(PhyReg.Reg);
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
*out << std::dec << (uint32_t)PhyReg.Reg;
}
return;
@@ -124,46 +124,41 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
}
}
static void PrintArg(fextl::stringstream* out, const IRListView*, FenceType Arg) {
*out << [Arg] {
switch (Arg) {
case FenceType::Load: return "Loads";
case FenceType::Store: return "Stores";
case FenceType::LoadStore: return "LoadStores";
case FenceType::Inst: return "Instruction";
}
return "<Unknown Fence Type>";
}();
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::FenceType Arg) {
if (Arg == IR::Fence_Load) {
*out << "Loads";
} else if (Arg == IR::Fence_Store) {
*out << "Stores";
} else if (Arg == IR::Fence_LoadStore) {
*out << "LoadStores";
} else {
*out << "<Unknown Fence Type>";
}
}
static void PrintArg(fextl::stringstream* out, const IRListView*, RoundMode Arg) {
*out << [Arg] {
switch (Arg) {
case RoundMode::Nearest: return "Nearest";
case RoundMode::NegInfinity: return "-Inf";
case RoundMode::PosInfinity: return "+Inf";
case RoundMode::TowardsZero: return "Towards Zero";
case RoundMode::Host: return "Host";
}
return "<Unknown Round Type>";
}();
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::RoundType Arg) {
switch (Arg) {
case FEXCore::IR::Round_Nearest: *out << "Nearest"; break;
case FEXCore::IR::Round_Negative_Infinity: *out << "-Inf"; break;
case FEXCore::IR::Round_Positive_Infinity: *out << "+Inf"; break;
case FEXCore::IR::Round_Towards_Zero: *out << "Towards Zero"; break;
case FEXCore::IR::Round_Host: *out << "Host"; break;
default: *out << "<Unknown Round Type>"; break;
}
}
static void PrintArg(fextl::stringstream* out, const IRListView*, SyscallFlags Arg) {
*out << [Arg] {
switch (Arg) {
case SyscallFlags::DEFAULT: return "Default";
case SyscallFlags::OPTIMIZETHROUGH: return "Optimize Through";
case SyscallFlags::NOSYNCSTATEONENTRY: return "No Sync State on Entry";
case SyscallFlags::NORETURN: return "No Return";
case SyscallFlags::NOSIDEEFFECTS: return "No Side Effects";
case SyscallFlags::NORETURNEDRESULT: return "No Returned Result";
}
return "<Unknown Syscall Flags>";
}();
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::SyscallFlags Arg) {
switch (Arg) {
case FEXCore::IR::SyscallFlags::DEFAULT: *out << "Default"; break;
case FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH: *out << "Optimize Through"; break;
case FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY: *out << "No Sync State on Entry"; break;
case FEXCore::IR::SyscallFlags::NORETURN: *out << "No Return"; break;
case FEXCore::IR::SyscallFlags::NOSIDEEFFECTS: *out << "No Side Effects"; break;
default: *out << "<Unknown Round Type>"; break;
}
}
static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorConstant Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::NamedVectorConstant Arg) {
*out << [Arg] {
// clang-format off
switch (Arg) {
@@ -191,22 +186,6 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorCon
return "movmskps_shift";
case NamedVectorConstant::NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE:
return "aeskeygenassist_swizzle";
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_0110B:
return "blendps_0110b";
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_0111B:
return "blendps_0111b";
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1001B:
return "blendps_1001b";
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1011B:
return "blendps_1011b";
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1101B:
return "blendps_1101b";
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1110B:
return "blendps_1110b";
case NamedVectorConstant::NAMED_VECTOR_MOVMASKB:
return "movmaskb";
case NamedVectorConstant::NAMED_VECTOR_MOVMASKB_UPPER:
return "movmaskb_upper";
case NamedVectorConstant::NAMED_VECTOR_ZERO:
return "vectorzero";
case NamedVectorConstant::NAMED_VECTOR_X87_ONE:
@@ -237,20 +216,9 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorCon
return "cvtmax_i32";
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I64:
return "cvtmax_i64";
case NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK:
return "f80_sign_mask";
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0:
return "sha1rnds_k0";
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1:
return "sha1rnds_k1";
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2:
return "sha1rnds_k2";
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3:
return "sha1rnds_k3";
case NamedVectorConstant::NAMED_VECTOR_MAX:
return "<Programming Error: Printing MAX value>";
default:
return "<Unknown Named Vector Constant>";
}
return "<Unknown Named Vector Constant>";
// clang-format on
}();
}
@@ -273,43 +241,36 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, IndexNamedVect
return "dppd_mask";
case IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW:
return "pblendw";
case INDEXED_NAMED_VECTOR_MAX:
return "<Programming Error: Printing MAX value>";
default:
return "<Unknown Indexed Named Vector Constant>";
}
return "<Unknown Indexed Named Vector Constant>";
// clang-format on
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, OpSize Arg) {
*out << [Arg] {
switch (Arg) {
case OpSize::iUnsized: return "Unsized";
case OpSize::i8Bit: return "i8";
case OpSize::i16Bit: return "i16";
case OpSize::i32Bit: return "i32";
case OpSize::i64Bit: return "i64";
case OpSize::f80Bit: return "f80";
case OpSize::i128Bit: return "i128";
case OpSize::i256Bit: return "i256";
case OpSize::iInvalid: return "Invalid";
}
return "<Unknown OpSize Type>";
}();
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::OpSize Arg) {
switch (Arg) {
case OpSize::i8Bit: *out << "i8"; break;
case OpSize::i16Bit: *out << "i16"; break;
case OpSize::i32Bit: *out << "i32"; break;
case OpSize::i64Bit: *out << "i64"; break;
case OpSize::i128Bit: *out << "i128"; break;
case OpSize::i256Bit: *out << "i256"; break;
case OpSize::f80Bit: *out << "f80"; break;
default: *out << "<Unknown OpSize Type>"; break;
}
}
static void PrintArg(fextl::stringstream* out, const IRListView*, FloatCompareOp Arg) {
*out << [Arg] {
switch (Arg) {
case FloatCompareOp::EQ: return "FEQ";
case FloatCompareOp::LT: return "FLT";
case FloatCompareOp::LE: return "FLE";
case FloatCompareOp::UNO: return "UNO";
case FloatCompareOp::NEQ: return "NEQ";
case FloatCompareOp::ORD: return "ORD";
}
return "<Unknown FloatCompareOp Type>";
}();
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::FloatCompareOp Arg) {
switch (Arg) {
case FloatCompareOp::EQ: *out << "FEQ"; break;
case FloatCompareOp::LT: *out << "FLT"; break;
case FloatCompareOp::LE: *out << "FLE"; break;
case FloatCompareOp::UNO: *out << "UNO"; break;
case FloatCompareOp::NEQ: *out << "NEQ"; break;
case FloatCompareOp::ORD: *out << "ORD"; break;
default: *out << "<Unknown OpSize Type>"; break;
}
}
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::BreakDefinition Arg) {
@@ -319,28 +280,23 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::B
*out << static_cast<uint32_t>(Arg.si_code) << "}";
}
static void PrintArg(fextl::stringstream* out, const IRListView*, ShiftType Arg) {
*out << [Arg] {
switch (Arg) {
case ShiftType::LSL: return "LSL";
case ShiftType::LSR: return "LSR";
case ShiftType::ASR: return "ASR";
case ShiftType::ROR: return "ROR";
}
return "<Unknown Shift Type>";
}();
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::ShiftType Arg) {
switch (Arg) {
case ShiftType::LSL: *out << "LSL"; break;
case ShiftType::LSR: *out << "LSR"; break;
case ShiftType::ASR: *out << "ASR"; break;
case ShiftType::ROR: *out << "ROR"; break;
default: *out << "<Unknown Shift Type>"; break;
}
}
static void PrintArg(fextl::stringstream* out, const IRListView*, BranchHint Arg) {
*out << [Arg] {
switch (Arg) {
case BranchHint::None: return "None";
case BranchHint::Call: return "Call";
case BranchHint::Return: return "Return";
case BranchHint::CheckTF: return "CheckTF";
}
return "<Unknown Branch Hint>";
}();
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::BranchHint Arg) {
switch (Arg) {
case BranchHint::None: *out << "None"; break;
case BranchHint::Call: *out << "Call"; break;
case BranchHint::Return: *out << "Return"; break;
default: *out << "<Unknown Branch Hint>"; break;
}
}
static void PrintArg(fextl::stringstream* out, const IRListView*, const std::array<uint8_t, 0x10>& Arg) {
@@ -359,8 +315,7 @@ void Dump(fextl::stringstream* out, const IRListView* IR) {
++CurrentIndent;
AddIndent();
*out << fextl::fmt::format("(%0) IRHeader %{}, #{:#x}, #{}, #{}\n", HeaderOp->Blocks.ID(), HeaderOp->OriginalRIP, HeaderOp->BlockCount,
HeaderOp->NumHostInstructions);
*out << fextl::fmt::format("(%0) IRHeader %{}, #{:#x}, #{}, #{}\n", HeaderOp->Blocks.ID(), HeaderOp->OriginalRIP, HeaderOp->BlockCount, HeaderOp->NumHostInstructions);
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
{
@@ -397,17 +352,17 @@ void Dump(fextl::stringstream* out, const IRListView* IR) {
auto PhyReg = PhysicalRegister(CodeNode);
if (!PhyReg.IsInvalid()) {
switch (PhyReg.AsRegClass()) {
case RegClass::GPR: *out << "(r"; break;
case RegClass::GPRFixed: *out << "(R"; break;
case RegClass::FPR: *out << "(v"; break;
case RegClass::FPRFixed: *out << "(V"; break;
case RegClass::Complex: *out << "(complex"; break;
case RegClass::Invalid: *out << "(invalid"; break;
switch (PhyReg.Class) {
case FEXCore::IR::GPRClass.Val: *out << "(r"; break;
case FEXCore::IR::GPRFixedClass.Val: *out << "(R"; break;
case FEXCore::IR::FPRClass.Val: *out << "(v"; break;
case FEXCore::IR::FPRFixedClass.Val: *out << "(V"; break;
case FEXCore::IR::ComplexClass.Val: *out << "(complex"; break;
case FEXCore::IR::InvalidClass.Val: *out << "(invalid"; break;
default: *out << "(unknown"; break;
}
if (PhyReg.AsRegClass() != RegClass::Invalid) {
*out << std::dec << uint32_t(PhyReg.Reg) << ")";
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
} else {
*out << ")";
}
+7 -7
View File
@@ -33,14 +33,14 @@ bool IsBlockExit(FEXCore::IR::IROps Op) {
}
}
RegClass IREmitter::WalkFindRegClass(Ref Node) {
FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
auto Class = GetOpRegClass(Node);
switch (Class) {
case RegClass::GPR:
case RegClass::FPR:
case RegClass::GPRFixed:
case RegClass::FPRFixed:
case RegClass::Invalid: return Class;
case GPRClass:
case FPRClass:
case GPRFixedClass:
case FPRFixedClass:
case InvalidClass: return Class;
default: break;
}
@@ -82,7 +82,7 @@ RegClass IREmitter::WalkFindRegClass(Ref Node) {
}
default: LOGMAN_MSG_A_FMT("Unhandled op type: {} {} in argument class validation", ToUnderlying(IROp->Op), GetOpName(Node)); break;
}
return RegClass::Invalid;
return InvalidClass;
}
void IREmitter::ResetWorkingList() {
+23 -68
View File
@@ -16,8 +16,13 @@
#include <string.h>
namespace FEXCore::IR {
class Pass;
class PassManager;
class IREmitter {
friend class FEXCore::IR::Pass;
friend class FEXCore::IR::PassManager;
public:
IREmitter(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, bool SupportsTSOImm9)
: DualListData {ThreadAllocator, 8 * 1024 * 1024}
@@ -46,12 +51,12 @@ public:
*
* @{ */
RegClass WalkFindRegClass(Ref Node);
FEXCore::IR::RegisterClassType WalkFindRegClass(Ref Node);
// These inlining helpers are used by IRDefines.inc so define first.
Ref InlineMem(OpSize Size, Ref Offset, MemOffsetType OffsetType, uint8_t& OffsetScale, bool TSO = false) {
uint64_t Imm {};
if (OffsetType != MemOffsetType::SXTX || !IsValueConstant(WrapNode(Offset), &Imm)) {
if (OffsetType != MEM_OFFSET_SXTX || !IsValueConstant(WrapNode(Offset), &Imm)) {
return Offset;
}
@@ -108,86 +113,36 @@ public:
IRPair<IROp_Jump> _Jump() {
return _Jump(InvalidNode);
}
IRPair<IROp_CondJump> _CondJump(Ref ssa0, CondClass cond = CondClass::NEQ) {
IRPair<IROp_CondJump> _CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
return _CondJump(ssa0, _Constant(0), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
}
IRPair<IROp_CondJump> _CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClass cond = CondClass::NEQ) {
IRPair<IROp_CondJump> _CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
}
// TODO: Work to remove this implicit sized Select implementation.
IRPair<IROp_Select> _Select(uint8_t Cond, Ref ssa0, Ref ssa1, Ref ssa2, Ref ssa3, IR::OpSize CompareSize = OpSize::iUnsized) {
if (CompareSize == OpSize::iUnsized) {
CompareSize = std::max(OpSize::i32Bit, std::max(GetOpSize(ssa0), GetOpSize(ssa1)));
}
IRPair<IROp_LoadContext> _LoadContextGPR(OpSize ByteSize, uint32_t Offset) {
return _LoadContext(ByteSize, RegClass::GPR, Offset);
return _Select(std::max(OpSize::i32Bit, std::max(GetOpSize(ssa2), GetOpSize(ssa3))), CompareSize, CondClassType {Cond}, ssa0, ssa1, ssa2, ssa3);
}
IRPair<IROp_LoadContext> _LoadContextFPR(OpSize ByteSize, uint32_t Offset) {
return _LoadContext(ByteSize, RegClass::FPR, Offset);
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = OpSize::i8Bit) {
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
IRPair<IROp_StoreContext> _StoreContextGPR(OpSize ByteSize, Ref Value, uint32_t Offset) {
return _StoreContext(ByteSize, RegClass::GPR, Value, Offset);
}
IRPair<IROp_StoreContext> _StoreContextFPR(OpSize ByteSize, Ref Value, uint32_t Offset) {
return _StoreContext(ByteSize, RegClass::FPR, Value, Offset);
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = OpSize::i8Bit) {
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
IRPair<IROp_LoadContextIndexed> _LoadContextGPRIndexed(Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
return _LoadContextIndexed(Index, ByteSize, BaseOffset, Stride, RegClass::GPR);
}
IRPair<IROp_LoadContextIndexed> _LoadContextFPRIndexed(Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
return _LoadContextIndexed(Index, ByteSize, BaseOffset, Stride, RegClass::FPR);
}
IRPair<IROp_StoreContextIndexed> _StoreContextGPRIndexed(Ref Value, Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
return _StoreContextIndexed(Value, Index, ByteSize, BaseOffset, Stride, RegClass::GPR);
}
IRPair<IROp_StoreContextIndexed> _StoreContextFPRIndexed(Ref Value, Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
return _StoreContextIndexed(Value, Index, ByteSize, BaseOffset, Stride, RegClass::FPR);
}
IRPair<IROp_LoadMem> _LoadMem(RegClass Class, OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
}
IRPair<IROp_LoadMem> _LoadMemGPR(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
return _LoadMem(RegClass::GPR, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
}
IRPair<IROp_LoadMem> _LoadMemGPR(OpSize Size, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
return _LoadMem(RegClass::GPR, Size, Addr, Offset, Align, OffsetType, OffsetScale);
}
IRPair<IROp_LoadMem> _LoadMemFPR(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
return _LoadMem(RegClass::FPR, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
}
IRPair<IROp_LoadMem> _LoadMemFPR(OpSize Size, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
return _LoadMem(RegClass::FPR, Size, Addr, Offset, Align, OffsetType, OffsetScale);
}
IRPair<IROp_StoreMem> _StoreMem(RegClass Class, OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
}
IRPair<IROp_StoreMem> _StoreMemGPR(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMem(RegClass::GPR, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
}
IRPair<IROp_StoreMem> _StoreMemGPR(OpSize Size, Ref Value, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
return _StoreMem(RegClass::GPR, Size, Value, Addr, Offset, Align, OffsetType, OffsetScale);
}
IRPair<IROp_StoreMem> _StoreMemFPR(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
return _StoreMem(RegClass::FPR, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
}
IRPair<IROp_StoreMem> _StoreMemFPR(OpSize Size, Ref Value, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
return _StoreMem(RegClass::FPR, Size, Value, Addr, Offset, Align, OffsetType, OffsetScale);
}
IRPair<IROp_StoreMemPair> _StoreMemPairGPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) {
return _StoreMemPair(RegClass::GPR, Size, Value1, Value2, Addr, Offset);
}
IRPair<IROp_StoreMemPair> _StoreMemPairFPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) {
return _StoreMemPair(RegClass::FPR, Size, Value1, Value2, Addr, Offset);
}
IRPair<IROp_Select> Select01(FEXCore::IR::OpSize CompareSize, CondClass Cond, OrderedNode* Cmp1, OrderedNode* Cmp2) {
IRPair<IROp_Select> Select01(FEXCore::IR::OpSize CompareSize, CondClassType Cond, OrderedNode* Cmp1, OrderedNode* Cmp2) {
return _Select(OpSize::i64Bit, CompareSize, Cond, Cmp1, Cmp2, _InlineConstant(1), _InlineConstant(0));
}
IRPair<IROp_Select> To01(FEXCore::IR::OpSize CompareSize, OrderedNode* Cmp1) {
return Select01(CompareSize, CondClass::NEQ, Cmp1, Constant(0));
return Select01(CompareSize, CondClassType {COND_NEQ}, Cmp1, Constant(0));
}
IRPair<IROp_NZCVSelect> _NZCVSelect01(CondClass Cond) {
IRPair<IROp_NZCVSelect> _NZCVSelect01(CondClassType Cond) {
return _NZCVSelect(OpSize::i64Bit, Cond, _InlineConstant(1), _InlineConstant(0));
}
@@ -300,7 +255,7 @@ public:
}
/** @} */
RegClass WalkFindRegClass(OrderedNodeWrapper ssa) {
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNodeWrapper ssa) {
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
return WalkFindRegClass(RealNode);
}
@@ -93,11 +93,11 @@ void IRValidation::Run(IREmitter* IREmit) {
// After RA, the destination needs to be assigned a register and class
auto PhyReg = PhysicalRegister(CodeNode);
const auto ExpectedClass = IR::GetRegClass(IROp->Op);
const auto AssignedClass = PhyReg.AsRegClass();
FEXCore::IR::RegisterClassType ExpectedClass = IR::GetRegClass(IROp->Op);
FEXCore::IR::RegisterClassType AssignedClass = FEXCore::IR::RegisterClassType {PhyReg.Class};
// If no register class was assigned
if (AssignedClass == IR::RegClass::Invalid) {
if (AssignedClass == IR::InvalidClass) {
HadError |= true;
Errors << "%" << ID << ": Had destination but with no register class assigned" << std::endl;
}
@@ -109,10 +109,10 @@ void IRValidation::Run(IREmitter* IREmit) {
}
// Assigned class wasn't the expected class and it is a non-complex op
if (AssignedClass != ExpectedClass && ExpectedClass != IR::RegClass::Complex) {
if (AssignedClass != ExpectedClass && ExpectedClass != IR::ComplexClass) {
HadWarning |= true;
Warnings << "%" << ID << ": Destination had register class " << uint32_t(AssignedClass) << " When register class "
<< uint32_t(ExpectedClass) << " Was expected" << std::endl;
Warnings << "%" << ID << ": Destination had register class " << AssignedClass.Val << " When register class "
<< ExpectedClass.Val << " Was expected" << std::endl;
}
}
}
@@ -2,20 +2,21 @@
/*
$info$
tags: ir|opts
desc: This is not used right now, possibly broken
$end_info$
*/
#include "FEXCore/Core/X86Enums.h"
#include "FEXCore/Utils/CompilerDefs.h"
#include "FEXCore/Utils/MathUtils.h"
#include "FEXCore/fextl/deque.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/deque.h>
#include <FEXCore/fextl/vector.h>
#include "Interface/IR/PassManager.h"
// Flag bit flags
#define FLAG_V (1U << 0)
@@ -61,36 +62,36 @@ struct FlagInfo {
return {.Raw = R};
}
bool Trivial() const {
bool Trivial() {
return Raw == 0;
}
unsigned Read() const {
unsigned Read() {
return Bits(0, 8);
}
unsigned Write() const {
unsigned Write() {
return Bits(8, 8);
}
bool CanEliminate() const {
bool CanEliminate() {
return Bits(16, 1);
}
bool Special() const {
bool Special() {
return Bits(63, 1);
}
IROps Replacement() const {
IROps Replacement() {
return (IROps)Bits(32, 16);
}
IROps ReplacementNoWrite() const {
IROps ReplacementNoWrite() {
return (IROps)Bits(48, 16);
}
private:
unsigned Bits(unsigned Start, unsigned Count) const {
unsigned Bits(unsigned Start, unsigned Count) {
return (Raw >> Start) & ((1u << Count) - 1);
}
};
@@ -153,44 +154,45 @@ public:
private:
FlagInfo Classify(IROp_Header* Node);
unsigned FlagsForCondClassType(CondClass Cond);
unsigned FlagForReg(unsigned Reg);
unsigned FlagsForCondClassType(CondClassType Cond);
bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp);
void FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode);
CondClass X86ToArmFloatCond(CondClass X86);
CondClassType X86ToArmFloatCond(CondClassType X86);
bool ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG);
void OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG);
};
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClass Cond) {
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) {
switch (Cond) {
case CondClass::AL: return 0;
case COND_AL: return 0;
case CondClass::MI:
case CondClass::PL: return FLAG_N;
case COND_MI:
case COND_PL: return FLAG_N;
case CondClass::EQ:
case CondClass::NEQ: return FLAG_Z;
case COND_EQ:
case COND_NEQ: return FLAG_Z;
case CondClass::UGE:
case CondClass::ULT: return FLAG_C;
case COND_UGE:
case COND_ULT: return FLAG_C;
case CondClass::VS:
case CondClass::VC:
case CondClass::FU:
case CondClass::FNU: return FLAG_V;
case COND_VS:
case COND_VC:
case COND_FU:
case COND_FNU: return FLAG_V;
case CondClass::UGT:
case CondClass::ULE: return FLAG_Z | FLAG_C;
case COND_UGT:
case COND_ULE: return FLAG_Z | FLAG_C;
case CondClass::SGE:
case CondClass::SLT:
case CondClass::FLU:
case CondClass::FGE: return FLAG_N | FLAG_V;
case COND_SGE:
case COND_SLT:
case COND_FLU:
case COND_FGE: return FLAG_N | FLAG_V;
case CondClass::SGT:
case CondClass::SLE:
case CondClass::FLEU:
case CondClass::FGT: return FLAG_N | FLAG_Z | FLAG_V;
case COND_SGT:
case COND_SLE:
case COND_FLEU:
case COND_FGT: return FLAG_N | FLAG_Z | FLAG_V;
default: LOGMAN_THROW_A_FMT(false, "unknown cond class type"); return FLAG_NZCV;
}
@@ -454,7 +456,7 @@ bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref
return true;
}
CondClass DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClass X86) {
CondClassType DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClassType X86) {
// Table of x86 condition codes that map to arm64 condition codes, in the
// sense that fcmp+axflag+branch(x86) is equivalent to fcmp+branch(arm).
//
@@ -463,12 +465,12 @@ CondClass DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClass X86) {
//
// SF/OF conditions are trivial and therefore shouldn't actually be generated
switch (X86) {
case CondClass::UGE /* A */: return CondClass::FGE /* GE */;
case CondClass::UGT /* AE */: return CondClass::FGT /* GT */;
case CondClass::ULT /* B */: return CondClass::SLT /* LT */;
case CondClass::ULE /* BE */: return CondClass::SLE /* LE */;
case CondClass::SLE /* LE */: return CondClass::SLE /* LE */;
default: return CondClass::AL;
case COND_UGE /* A */: return {COND_FGE} /* GE */;
case COND_UGT /* AE */: return {COND_FGT} /* GT */;
case COND_ULT /* B */: return {COND_SLT} /* LT */;
case COND_ULE /* BE */: return {COND_SLE} /* LE */;
case COND_SLE /* LE */: return {COND_SLE} /* LE */;
default: return {COND_AL};
}
}
@@ -483,8 +485,8 @@ void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView&
auto Prev = CurrentIR.GetOp<IR::IROp_Header>(PrevWrap);
if (Prev->Op == OP_AXFLAG) {
// Pattern match a branch fed by AXFLAG.
CondClass ArmCond = X86ToArmFloatCond(Op->Cond);
if (ArmCond == CondClass::AL) {
CondClassType ArmCond = X86ToArmFloatCond(Op->Cond);
if (ArmCond == COND_AL) {
return;
}
@@ -493,7 +495,7 @@ void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView&
// Pattern match a branch fed by a compare. We could also handle bit tests
// here, but tbz/tbnz has a limited offset range which we don't have a way to
// deal with yet. Let's hope that's not a big deal.
if (!(Op->Cond == CondClass::NEQ || Op->Cond == CondClass::EQ) || (Prev->Size < OpSize::i32Bit)) {
if (!(Op->Cond == COND_NEQ || Op->Cond == COND_EQ) || (Prev->Size < OpSize::i32Bit)) {
return;
}
@@ -627,9 +629,9 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
}
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
const auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
const auto& Predecessors = CFG.Get(ID)->Predecessors;
auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
bool Full = false;
auto Predecessors = CFG.Get(ID)->Predecessors;
if (Predecessors.empty()) {
// Conservatively assume there was full parity before the start block
@@ -12,7 +12,6 @@ $end_info$
#include "Interface/IR/Passes.h"
#include "Interface/Core/CPUID.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/vector.h>
@@ -23,7 +22,7 @@ using namespace FEXCore;
namespace FEXCore::IR {
namespace {
struct RegisterClassData {
struct RegisterClass {
uint32_t Available;
uint32_t Count;
@@ -33,9 +32,9 @@ namespace {
Ref RegToSSA[32];
};
IR::RegClass GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
const auto Class = IR::GetRegClass(IROp->Op);
if (Class != IR::RegClass::Complex) {
IR::RegisterClassType GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
IR::RegisterClassType Class = IR::GetRegClass(IROp->Op);
if (Class != IR::ComplexClass) {
return Class;
}
@@ -47,7 +46,7 @@ namespace {
case IR::OP_LOADMEM:
case IR::OP_LOADMEMTSO: return IROp->C<IR::IROp_LoadMem>()->Class;
case IR::OP_FILLREGISTER: return IROp->C<IR::IROp_FillRegister>()->Class;
default: return IR::RegClass::Invalid;
default: return IR::InvalidClass;
}
};
} // Anonymous namespace
@@ -57,11 +56,11 @@ public:
explicit ConstrainedRAPass(const FEXCore::CPUIDEmu* CPUID)
: CPUID {CPUID} {}
void Run(IREmitter* IREmit) override;
void AddRegisters(IR::RegClass Class, uint32_t RegisterCount) override;
void AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) override;
bool TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header* IROp);
private:
RegisterClassData Classes[IR::NumClasses];
RegisterClass Classes[IR::NumClasses];
IREmitter* IREmit;
IRListView* IR;
@@ -102,7 +101,7 @@ private:
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value];
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Node must have been spilled");
const auto RegClass = GetRegClassFromNode(IR, IROp);
RegisterClassType RegClass = GetRegClassFromNode(IR, IROp);
return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, SlotPlusOne - 1, RegClass);
};
@@ -121,7 +120,7 @@ private:
return Op != OP_INLINECONSTANT && Op != OP_INLINEENTRYPOINTOFFSET;
};
RegisterClassData* GetClass(PhysicalRegister Reg) {
RegisterClass* GetClass(PhysicalRegister Reg) {
return &Classes[Reg.Class];
};
@@ -134,13 +133,13 @@ private:
LOGMAN_THROW_A_FMT(ID < SSAToReg.size(), "Only old nodes looked up");
PhysicalRegister Reg = SSAToReg[ID];
RegisterClassData* Class = GetClass(Reg);
RegisterClass* Class = GetClass(Reg);
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Node;
};
void FreeReg(PhysicalRegister Reg) {
RegisterClassData* Class = GetClass(Reg);
RegisterClass* Class = GetClass(Reg);
uint32_t RegBits = GetRegBits(Reg);
LOGMAN_THROW_A_FMT(!(Class->Available & RegBits), "Register double-free");
@@ -188,22 +187,22 @@ private:
};
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) {
uint8_t FlagOffset = Classes[FEXCore::ToUnderlying(RegClass::GPRFixed)].Count - 2;
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
if (IROp->Op == OP_STOREREGISTER) {
return PhysicalRegister(Node);
} else if (IROp->Op == OP_LOADPF || IROp->Op == OP_STOREPF) {
return PhysicalRegister {RegClass::GPRFixed, FlagOffset};
return PhysicalRegister {GPRFixedClass, FlagOffset};
} else if (IROp->Op == OP_LOADAF || IROp->Op == OP_STOREAF) {
return PhysicalRegister {RegClass::GPRFixed, uint8_t(FlagOffset + 1)};
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
} else {
const IROp_LoadRegister* Op = IROp->C<IR::IROp_LoadRegister>();
LOGMAN_THROW_A_FMT(Op->Class == RegClass::GPR || Op->Class == RegClass::FPR, "SRA classes");
if (Op->Class == RegClass::FPR) {
return PhysicalRegister {RegClass::FPRFixed, uint8_t(Op->Reg)};
LOGMAN_THROW_A_FMT(Op->Class == GPRClass || Op->Class == FPRClass, "SRA classes");
if (Op->Class == FPRClass) {
return PhysicalRegister {FPRFixedClass, (uint8_t)Op->Reg};
} else {
return PhysicalRegister {RegClass::GPRFixed, uint8_t(Op->Reg)};
return PhysicalRegister {GPRFixedClass, (uint8_t)Op->Reg};
}
}
};
@@ -268,7 +267,7 @@ private:
SourceIndex = SourcesNextUses.size();
}
void SpillReg(RegisterClassData* Class, IROp_CodeBlock* Block, IROp_Header* Exclude) {
void SpillReg(RegisterClass* Class, IROp_CodeBlock* Block, IROp_Header* Exclude) {
// We're about to use next-use information, so calculate it.
if (!AnySpilled) {
CalculateNextUses(Block, Exclude);
@@ -319,7 +318,7 @@ private:
// If we already spilled the Candidate, we don't need to spill again.
// Similarly, if we can rematerialize the instruction, we don't spill it.
if (!Spilled && Header->Op != OP_CONSTANT) {
LOGMAN_THROW_A_FMT(Reg.AsRegClass() == GetRegClassFromNode(IR, Header), "Consistent");
LOGMAN_THROW_A_FMT(Reg.Class == GetRegClassFromNode(IR, Header), "Consistent");
// SpillSlots allocation is deferred.
if (SpillSlots.empty()) {
@@ -330,7 +329,7 @@ private:
uint32_t Slot = IR->GetHeader()->SpillSlots++;
// We must map here in case we're spilling something we shuffled.
auto SpillOp = IREmit->_SpillRegister(OrderedNodeWrapper::FromImmediate(Reg.Raw), Slot, Reg.AsRegClass());
auto SpillOp = IREmit->_SpillRegister(OrderedNodeWrapper::FromImmediate(Reg.Raw), Slot, RegisterClassType {Reg.Class});
SpillOp.first->Header.Size = Header->Size;
SpillOp.first->Header.ElementSize = Header->ElementSize;
SpillSlots[Value] = Slot + 1;
@@ -342,7 +341,7 @@ private:
};
void RemapReg(Ref Node, PhysicalRegister Reg) {
RegisterClassData* Class = GetClass(Reg);
RegisterClass* Class = GetClass(Reg);
Class->RegToSSA[Reg.Reg] = Node;
uint32_t Index = IR->GetID(Node).Value;
@@ -353,7 +352,7 @@ private:
// Record a given assignment of register Reg to Node.
void SetReg(Ref Node, PhysicalRegister Reg) {
RegisterClassData* Class = GetClass(Reg);
RegisterClass* Class = GetClass(Reg);
uint32_t RegBits = GetRegBits(Reg);
LOGMAN_THROW_A_FMT((Class->Available & RegBits) == RegBits, "Precondition");
@@ -371,7 +370,7 @@ private:
// Prioritize preferred registers.
if (Node < PreferredReg.size()) {
if (PhysicalRegister Reg = PreferredReg[Node]; !Reg.IsInvalid()) {
RegisterClassData* Class = GetClass(Reg);
RegisterClass* Class = GetClass(Reg);
uint32_t RegBits = GetRegBits(Reg);
if ((Class->Available & RegBits) == RegBits) {
@@ -384,10 +383,10 @@ private:
// Try to handle tied registers. This can fail, the JIT will insert moves.
if (int TiedIdx = IR::TiedSource(IROp->Op); TiedIdx >= 0) {
auto Reg = PhysicalRegister(IROp->Args[TiedIdx]);
RegisterClassData* Class = GetClass(Reg);
RegisterClass* Class = GetClass(Reg);
uint32_t RegBits = GetRegBits(Reg);
if (Reg.AsRegClass() != RegClass::GPRFixed && Reg.AsRegClass() != RegClass::FPRFixed && (Class->Available & RegBits) == RegBits) {
if (Reg.Class != GPRFixedClass && Reg.Class != FPRFixedClass && (Class->Available & RegBits) == RegBits) {
SetReg(CodeNode, Reg);
return;
}
@@ -395,7 +394,7 @@ private:
// Try to coalesce reserved pairs. Just a heuristic to remove some moves.
if (IROp->Op == OP_ALLOCATEGPR && IROp->C<IROp_AllocateGPR>()->ForPair) {
uint32_t Available = Classes[FEXCore::ToUnderlying(RegClass::GPR)].Available;
uint32_t Available = Classes[GPRClass].Available;
// Only choose base register R if R and R + 1 are both free
Available &= (Available >> 1);
@@ -406,20 +405,20 @@ private:
if (Available) {
unsigned Reg = std::countr_zero(Available);
SetReg(CodeNode, PhysicalRegister(RegClass::GPR, Reg));
SetReg(CodeNode, PhysicalRegister(GPRClass, Reg));
return;
}
} else if (IROp->Op == OP_ALLOCATEGPRAFTER) {
uint32_t Available = Classes[FEXCore::ToUnderlying(RegClass::GPR)].Available;
uint32_t Available = Classes[GPRClass].Available;
auto After = PhysicalRegister(IROp->Args[0]);
if ((After.Reg & 1) == 0 && Available & (1ull << (After.Reg + 1))) {
SetReg(CodeNode, PhysicalRegister(RegClass::GPR, After.Reg + 1));
SetReg(CodeNode, PhysicalRegister(GPRClass, After.Reg + 1));
return;
}
}
RegClass ClassType = GetRegClassFromNode(IR, IROp);
RegisterClassData* Class = &Classes[FEXCore::ToUnderlying(ClassType)];
RegisterClassType ClassType = GetRegClassFromNode(IR, IROp);
RegisterClass* Class = &Classes[ClassType];
// Spill to make room in the register file.
if (!Class->Available) {
@@ -434,10 +433,10 @@ private:
};
};
void ConstrainedRAPass::AddRegisters(IR::RegClass Class, uint32_t RegisterCount) {
void ConstrainedRAPass::AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) {
LOGMAN_THROW_A_FMT(RegisterCount <= 31, "Up to 31 regs supported");
Classes[FEXCore::ToUnderlying(Class)].Count = RegisterCount;
Classes[Class].Count = RegisterCount;
}
inline bool KillMove(IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref CodeNode) {
@@ -531,7 +530,7 @@ bool ConstrainedRAPass::TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header*
const auto Result = CPUID->RunFunction(ConstantFunction, 0 /* leaf */);
IREmit->SetWriteCursorBefore(CodeNode);
IREmit->_Fence(IR::FenceType::Inst);
IREmit->_Fence({FEXCore::IR::Fence_Inst});
IREmit->_Constant(Result.eax).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
IREmit->_Constant(Result.ebx).Node->Reg = PhysicalRegister(Op->OutEBX).Raw;
IREmit->_Constant(Result.ecx).Node->Reg = PhysicalRegister(Op->OutECX).Raw;
@@ -664,7 +663,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
// Static registers must be consistent at SRA load/store. Evict to ensure.
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
auto Reg = DecodeSRAReg(IROp, CodeNode);
RegisterClassData* Class = &Classes[Reg.Class];
RegisterClass* Class = &Classes[Reg.Class];
if (!(Class->Available & (1u << Reg.Reg))) {
Ref Old = Class->RegToSSA[Reg.Reg];
@@ -679,7 +678,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
Ref Copy;
if (Reg.AsRegClass() == RegClass::FPRFixed) {
if (Reg.Class == FPRFixedClass) {
IROp_Header* Header = IR->GetOp<IROp_Header>(Old);
Copy = IREmit->_VMov(Header->Size, OrderedNodeWrapper::FromImmediate(Reg.Raw));
} else {
@@ -12,11 +12,11 @@ $end_info$
#include <stdint.h>
namespace FEXCore::IR {
enum class RegClass : uint32_t;
struct RegisterClassType;
class RegisterAllocationPass : public FEXCore::IR::Pass {
public:
virtual void AddRegisters(RegClass Class, uint32_t RegisterCount) = 0;
virtual void AddRegisters(FEXCore::IR::RegisterClassType Class, uint32_t RegisterCount) = 0;
// Number of GPRs usable for pairs at start of GPR set. Must be even.
uint32_t PairRegs;
@@ -158,9 +158,7 @@ public:
: Features(Features)
, GPROpSize(GPROpSize) {
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(StrictReducedPrecision, X87STRICTREDUCEDPRECISION);
ReducedPrecisionMode = ReducedPrecision;
StrictReducedPrecisionMode = StrictReducedPrecision;
}
void Run(IREmitter* Emit) override;
@@ -168,12 +166,10 @@ private:
const FEXCore::HostFeatures& Features;
const OpSize GPROpSize;
bool ReducedPrecisionMode;
bool StrictReducedPrecisionMode;
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
// Helpers
Ref RotateRight8(uint32_t V, Ref Amount);
Ref SilenceNaN(Ref Value);
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
Ref AddrNode = IR->GetNode(Op->Addr);
@@ -182,18 +178,18 @@ private:
MemOffsetType OffsetType = Op->OffsetType;
uint8_t OffsetScale = Op->OffsetScale;
IREmit->_StoreMemFPR(OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
// Store the Upper part of the register (the remaining 2 bytes) into memory.
AddressMode A {.Base = AddrNode,
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
.Offset = 8,
.IndexType = MemOffsetType::SXTX,
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = OffsetScale,
.Offset = 8,
.AddrSize = OpSize::i64Bit};
A = SelectAddressMode(IREmit, A, GPROpSize, Features.SupportsTSOImm9, false, false, OpSize::i16Bit);
IREmit->_StoreMemGPR(OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale);
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MEM_OFFSET_SXTX, A.IndexScale);
}
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
@@ -208,10 +204,7 @@ private:
case OpSize::i32Bit:
case OpSize::i64Bit: {
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
StackNode = SilenceNaN(StackNode);
}
IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
break;
}
@@ -219,7 +212,7 @@ private:
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
AddressMode A {.Base = AddrNode,
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
.IndexType = MemOffsetType::SXTX,
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = OffsetScale,
.AddrSize = OpSize::i64Bit};
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
@@ -242,17 +235,13 @@ private:
MemOffsetType OffsetType = Op->OffsetType;
uint8_t OffsetScale = Op->OffsetScale;
if ((!ReducedPrecisionMode || StrictReducedPrecisionMode) && Op->StoreSize != OpSize::f80Bit) {
StackNode = SilenceNaN(StackNode);
}
switch (Op->StoreSize) {
case OpSize::i32Bit: {
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
[[fallthrough]];
}
case OpSize::i64Bit: {
IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
break;
}
@@ -266,6 +255,18 @@ private:
}
}
// Helper to check if a Ref is a Zero constant
bool IsZero(Ref Node) {
auto Header = IR->GetOp<IR::IROp_Header>(Node);
if (Header->Op != OP_CONSTANT) {
return false;
}
auto Const = Header->C<IROp_Constant>();
return Const->Constant == 0;
}
// Handles a Unary operation.
// Takes the op we are handling, the Node for the reduced precision case and the node for the normal case.
// Depending on the type of Op64, we might need to pass a couple of extra constant arguments, this happens
@@ -278,7 +279,7 @@ private:
// Top Management Helpers
/// Set the valid tag for Value as valid (if Valid is true), or invalid (if Valid is false).
void SetX87ValidTag(uint8_t Offset, bool Valid);
void SetX87ValidTag(Ref Value, bool Valid);
// Generates slow code to load/store a value from an offset from the top of the stack
Ref LoadStackValueAtOffset_Slow(uint8_t Offset = 0);
void StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset = 0, bool SetValid = true);
@@ -293,12 +294,11 @@ private:
void MigrateToSlowPathIf(bool ShouldMigrate);
// Top Cache Management
Ref GetTopWithCache_Slow();
Ref GetOffsetTopWithCache_Slow(uint8_t Offset, bool Reverse = false);
Ref GetOffsetTopAddressWithCache_Slow(uint8_t Offset);
Ref GetOffsetTopWithCache_Slow(uint8_t Offset);
void SetTopWithCache_Slow(Ref Value);
Ref GetX87ValidTag_Slow(uint8_t Offset);
// Resets fields to initial values
void Reset();
void Reset(bool AlsoSlowPath = true);
struct StackMemberInfo {
StackMemberInfo() {}
@@ -330,7 +330,7 @@ private:
FixedSizeStack<StackMemberInfo> StackData;
void InvalidateCaches();
void InvalidateCachedRegs();
void InvalidateTopOffsetCache();
// Path Migration helper management
std::optional<StackMemberInfo> MigrateToSlowPath_IfInvalid(uint8_t Offset = 0);
@@ -345,19 +345,7 @@ private:
// Cached value for Top
// If slowpath is false, then TopCache is nullptr.
bool FlushTopPending = false;
std::array<bool, 8> FlushValuesPending {};
bool FlushValidPending = false;
void FlushCachedRegs();
Ref GetFTW();
Ref FTWCached {};
std::array<Ref, 8> TopOffsetCache {};
std::array<Ref, 8> TopOffsetAddressCache {};
std::array<Ref, 8> TopValueCache {};
std::array<StackSlot, 8> TopValidCache {};
// Are we on the slow path?
// Once we enter the slow path, we never come out.
// This just simplifies the code atm. If there's a need to return to the fast path in the future
@@ -371,21 +359,18 @@ private:
};
inline void X87StackOptimization::InvalidateCaches() {
InvalidateCachedRegs();
InvalidateTopOffsetCache();
ConstantPool.fill(nullptr);
}
inline void X87StackOptimization::InvalidateCachedRegs() {
FlushCachedRegs();
FTWCached = {};
inline void X87StackOptimization::InvalidateTopOffsetCache() {
TopOffsetCache.fill(nullptr);
TopOffsetAddressCache.fill(nullptr);
TopValueCache.fill(nullptr);
TopValidCache.fill(StackSlot::UNUSED);
}
inline void X87StackOptimization::Reset() {
SlowPath = false;
inline void X87StackOptimization::Reset(bool AlsoSlowPath) {
if (AlsoSlowPath) {
SlowPath = false;
}
StackData.clear();
InvalidateCaches();
}
@@ -405,25 +390,20 @@ inline Ref X87StackOptimization::GetConstant(ssize_t Offset) {
inline void X87StackOptimization::MigrateToSlowPathIf(bool ShouldMigrate) {
if (ShouldMigrate && !SlowPath) {
SynchronizeStackValues();
StackData.clear();
Reset(false); // Reset everything but no need to change slowpath
SlowPath = true;
}
}
inline Ref X87StackOptimization::GetTopWithCache_Slow() {
if (!TopOffsetCache[0]) {
TopOffsetCache[0] = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
TopOffsetCache[0] =
IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
}
return TopOffsetCache[0];
}
inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset, bool Reverse) {
if (Reverse) {
Offset = 8 - Offset;
}
Offset &= 7;
inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset) {
if (TopOffsetCache[Offset]) {
return TopOffsetCache[Offset];
}
@@ -438,60 +418,38 @@ inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset, bool
return OffsetTop;
}
inline Ref X87StackOptimization::GetOffsetTopAddressWithCache_Slow(uint8_t Offset) {
if (TopOffsetAddressCache[Offset]) {
return TopOffsetAddressCache[Offset];
}
Ref OffsetRef = GetOffsetTopWithCache_Slow(Offset);
TopOffsetAddressCache[Offset] = IREmit->_FormContextAddress(OpSize::i64Bit, OffsetRef, 16);
return TopOffsetAddressCache[Offset];
}
inline void X87StackOptimization::SetTopWithCache_Slow(Ref Value) {
InvalidateCachedRegs();
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
InvalidateTopOffsetCache();
TopOffsetCache[0] = Value;
FlushTopPending = true;
}
inline Ref X87StackOptimization::GetFTW() {
if (!FTWCached) {
FTWCached = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
}
return FTWCached;
}
inline void X87StackOptimization::SetX87ValidTag(uint8_t Offset, bool Valid) {
TopValidCache[Offset] = Valid ? StackSlot::VALID : StackSlot::INVALID;
FlushValidPending = true;
inline void X87StackOptimization::SetX87ValidTag(Ref Value, bool Valid) {
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), Value);
Ref NewAbridgedFTW = Valid ? IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RegMask) : IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RegMask);
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
}
inline Ref X87StackOptimization::GetX87ValidTag_Slow(uint8_t Offset) {
switch (TopValidCache[Offset]) {
case StackSlot::UNUSED:
return IREmit->_And(OpSize::i32Bit, IREmit->_Lshr(OpSize::i32Bit, GetFTW(), GetOffsetTopWithCache_Slow(Offset)), GetConstant(1));
case StackSlot::INVALID: return GetConstant(0);
case StackSlot::VALID: return GetConstant(1);
}
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
return IREmit->_And(OpSize::i32Bit, IREmit->_Lshr(OpSize::i32Bit, AbridgedFTW, GetOffsetTopWithCache_Slow(Offset)), GetConstant(1));
}
inline Ref X87StackOptimization::LoadStackValueAtOffset_Slow(uint8_t Offset) {
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(Offset);
auto Size = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
if (!TopValueCache[Offset]) {
TopValueCache[Offset] = IREmit->_LoadMemFPR(Size, TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
}
return TopValueCache[Offset];
return IREmit->_LoadContextIndexed(GetOffsetTopWithCache_Slow(Offset), ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
MMBaseOffset(), 16, FPRClass);
}
inline void X87StackOptimization::StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset, bool SetValid) {
TopValueCache[Offset] = Value;
FlushValuesPending[Offset] = true;
OrderedNode* TopOffset = GetOffsetTopWithCache_Slow(Offset);
// store
IREmit->_StoreContextIndexed(Value, TopOffset, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit, MMBaseOffset(), 16, FPRClass);
// mark it valid
// In some cases we might already know it has been previously set as valid so we don't need to do it again
if (SetValid) {
SetX87ValidTag(Offset, true);
SetX87ValidTag(TopOffset, true);
}
}
@@ -499,15 +457,6 @@ inline Ref X87StackOptimization::RotateRight8(uint32_t V, Ref Amount) {
return IREmit->_Lshr(OpSize::i32Bit, GetConstant(V | (V << 8)), Amount);
}
inline Ref X87StackOptimization::SilenceNaN(Ref Value) {
Ref GPRValue = IREmit->_VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, Value, 0);
IREmit->_FCmp(OpSize::i64Bit, Value, Value); // Comparison with itself should set VS if nan
Ref QuietNaNGPR = IREmit->_Or(OpSize::i64Bit, GPRValue, IREmit->_Constant(0x0008000000000000ULL));
Ref SilencedValue = IREmit->_VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, QuietNaNGPR);
return IREmit->_NZCVSelectV(OpSize::i64Bit, CondClass::VS, SilencedValue, Value);
}
inline std::optional<X87StackOptimization::StackMemberInfo> X87StackOptimization::MigrateToSlowPath_IfInvalid(uint8_t Offset) {
const auto& [Valid, StackMember] = StackData.top(Offset);
MigrateToSlowPathIf(Valid != StackSlot::VALID);
@@ -592,100 +541,25 @@ void X87StackOptimization::HandleBinopStack(IROps Op64, bool VFOp64, IROps Op80,
inline void X87StackOptimization::UpdateTopForPop_Slow() {
// Pop the top of the x87 stack
GetOffsetTopWithCache_Slow(1);
std::rotate(TopOffsetCache.begin(), std::next(TopOffsetCache.begin()), TopOffsetCache.end());
std::rotate(TopOffsetAddressCache.begin(), std::next(TopOffsetAddressCache.begin()), TopOffsetAddressCache.end());
std::rotate(TopValueCache.begin(), std::next(TopValueCache.begin()), TopValueCache.end());
std::rotate(FlushValuesPending.begin(), std::next(FlushValuesPending.begin()), FlushValuesPending.end());
std::rotate(TopValidCache.begin(), std::next(TopValidCache.begin()), TopValidCache.end());
FlushTopPending = true;
auto* TopOffset = GetTopWithCache_Slow();
TopOffset = IREmit->Add(OpSize::i32Bit, TopOffset, 1);
TopOffset = IREmit->_And(OpSize::i32Bit, TopOffset, GetConstant(7));
SetTopWithCache_Slow(TopOffset);
}
inline void X87StackOptimization::UpdateTopForPush_Slow() {
// Pop the top of the x87 stack
GetOffsetTopWithCache_Slow(1, true);
std::rotate(TopOffsetCache.begin(), std::prev(TopOffsetCache.end()), TopOffsetCache.end());
std::rotate(TopOffsetAddressCache.begin(), std::prev(TopOffsetAddressCache.end()), TopOffsetAddressCache.end());
std::rotate(TopValueCache.begin(), std::prev(TopValueCache.end()), TopValueCache.end());
std::rotate(FlushValuesPending.begin(), std::prev(FlushValuesPending.end()), FlushValuesPending.end());
std::rotate(TopValidCache.begin(), std::prev(TopValidCache.end()), TopValidCache.end());
FlushTopPending = true;
}
void X87StackOptimization::FlushCachedRegs() {
if (FlushTopPending) {
IREmit->_StoreContextGPR(OpSize::i8Bit, TopOffsetCache[0], offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
FlushTopPending = false;
}
auto Size = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (size_t i = 0; i < FlushValuesPending.size(); i++) {
if (FlushValuesPending[i]) {
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(i);
IREmit->_StoreMemFPR(Size, TopValueCache[i], TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
// store
FlushValuesPending[i] = false;
}
}
if (FlushValidPending) {
uint8_t ValidMask = 0;
uint8_t InvalidMask = 0;
for (auto It = TopValidCache.rbegin(); It != TopValidCache.rend(); It++) {
ValidMask <<= 1;
InvalidMask <<= 1;
if (*It == StackSlot::VALID) {
ValidMask |= 1;
} else if (*It == StackSlot::INVALID) {
InvalidMask |= 1;
}
}
if (ValidMask || InvalidMask) {
Ref NewFTW = [&]() {
if (ValidMask == 0xff || InvalidMask == 0xff) {
// If InvalidMask == 0xff then ValidMask = 0
return GetConstant(ValidMask);
} else {
Ref NewFTW = GetFTW();
Ref RotAmount {};
if (std::popcount(ValidMask) == 1) {
uint8_t BitIdx = std::countr_zero(ValidMask);
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), GetOffsetTopWithCache_Slow(BitIdx));
NewFTW = IREmit->_Or(OpSize::i32Bit, NewFTW, RegMask);
} else if (ValidMask) {
RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), GetTopWithCache_Slow());
// perform a rotate right on mask by top
NewFTW = IREmit->_Or(OpSize::i32Bit, NewFTW, RotateRight8(ValidMask, RotAmount));
}
if (std::popcount(InvalidMask) == 1) {
uint8_t BitIdx = std::countr_zero(InvalidMask);
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), GetOffsetTopWithCache_Slow(BitIdx));
NewFTW = IREmit->_Andn(OpSize::i32Bit, NewFTW, RegMask);
} else if (InvalidMask) {
if (!RotAmount) {
RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), GetTopWithCache_Slow());
}
NewFTW = IREmit->_Andn(OpSize::i32Bit, NewFTW, RotateRight8(InvalidMask, RotAmount));
}
return NewFTW;
}
}();
IREmit->_StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
FTWCached = NewFTW;
}
FlushValidPending = false;
}
auto* TopOffset = GetTopWithCache_Slow();
TopOffset = IREmit->Sub(OpSize::i32Bit, TopOffset, 1);
TopOffset = IREmit->_And(OpSize::i32Bit, TopOffset, GetConstant(7));
SetTopWithCache_Slow(TopOffset);
}
// We synchronize stack values in a few occasions but one of the most important of those,
// is when we move from fast to a slow path and need to make sure that the context is properly
// written.
Ref X87StackOptimization::SynchronizeStackValues() {
if (SlowPath) {
if (SlowPath) { // Nothing to do here.
return GetTopWithCache_Slow();
}
@@ -694,7 +568,8 @@ Ref X87StackOptimization::SynchronizeStackValues() {
const auto TopOffset = StackData.TopOffset;
if (TopOffset != 0) {
Ref NewTop = GetOffsetTopWithCache_Slow(TopOffset, true);
auto* OrigTop = GetTopWithCache_Slow();
Ref NewTop = IREmit->_And(OpSize::i32Bit, IREmit->Sub(OpSize::i32Bit, OrigTop, TopOffset), GetConstant(0x7));
SetTopWithCache_Slow(NewTop);
}
StackData.TopOffset = 0;
@@ -706,22 +581,51 @@ Ref X87StackOptimization::SynchronizeStackValues() {
for (size_t i = 0; i < StackData.size; ++i) {
const auto& [Valid, StackMember] = StackData.top(i);
if (Valid == StackSlot::UNUSED) {
continue;
}
Ref TopIndex = GetOffsetTopWithCache_Slow(i);
if (Valid == StackSlot::VALID) {
StoreStackValueAtOffset_Slow(StackMember.StackDataNode, i, false);
IREmit->_StoreContextIndexed(StackMember.StackDataNode, TopIndex, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
MMBaseOffset(), 16, FPRClass);
}
}
{ // Set valid tags
uint8_t ValidMask = StackData.getValidMask();
uint8_t InvalidMask = StackData.getInvalidMask();
for (auto& Elem : TopValidCache) {
Elem = (ValidMask & 1) ? StackSlot::VALID : ((InvalidMask & 1) ? StackSlot::INVALID : StackSlot::UNUSED);
ValidMask >>= 1;
InvalidMask >>= 1;
uint8_t Mask = StackData.getValidMask();
if (Mask == 0xff) {
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(Mask), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
} else if (Mask != 0) {
if (std::popcount(Mask) == 1) {
uint8_t BitIdx = __builtin_ctz(Mask);
SetX87ValidTag(GetOffsetTopWithCache_Slow(BitIdx), true);
} else {
// perform a rotate right on mask by top
auto* TopValue = GetTopWithCache_Slow();
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref NewAbridgedFTW = IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
}
}
}
{ // Set invalid tags
uint8_t Mask = StackData.getInvalidMask();
if (Mask == 0xff) {
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
} else if (Mask != 0) {
if (std::popcount(Mask)) {
uint8_t BitIdx = __builtin_ctz(Mask);
SetX87ValidTag(GetOffsetTopWithCache_Slow(BitIdx), false);
} else {
// Same rotate right as above but this time on the invalid mask
auto* TopValue = GetTopWithCache_Slow();
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref NewAbridgedFTW = IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
}
}
FlushValidPending = true;
}
return TopValue;
}
@@ -911,7 +815,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
case OP_INITSTACK: {
StackData.clear();
InvalidateCachedRegs();
break;
}
@@ -921,14 +824,18 @@ void X87StackOptimization::Run(IREmitter* Emit) {
if (Offset != 0xff) { // invalidate single offset
if (SlowPath) {
SetX87ValidTag(Offset, false);
auto* TopValue = GetTopWithCache_Slow();
if (Offset != 0) {
auto* Mask = GetConstant(7);
TopValue = IREmit->_And(OpSize::i32Bit, IREmit->Add(OpSize::i32Bit, TopValue, Offset), Mask);
}
SetX87ValidTag(TopValue, false);
} else {
StackData.setTagInvalid(Offset);
}
} else { // invalidate all
if (SlowPath) {
TopValidCache.fill(StackSlot::INVALID);
FlushValidPending = true;
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
} else {
for (size_t i = 0; i < StackData.size; i++) {
StackData.setTagInvalid(i);
@@ -1014,8 +921,8 @@ void X87StackOptimization::Run(IREmitter* Emit) {
// or similar. As long as the source size and dest size are one and the same.
// This will avoid any conversions between source and stack element size and conversion back.
if (!SlowPath && Value->Source && Value->Source->Size == Op->StoreSize && Value->InterpretAsFloat) {
const auto ClassType = Value->InterpretAsFloat ? RegClass::FPR : RegClass::GPR;
IREmit->_StoreMem(ClassType, Op->StoreSize, Value->Source->Node, AddrNode, Offset, Align, OffsetType, OffsetScale);
IREmit->_StoreMem(Value->InterpretAsFloat ? FPRClass : GPRClass, Op->StoreSize, Value->Source->Node, AddrNode, Offset, Align,
OffsetType, OffsetScale);
break;
}
@@ -1046,7 +953,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
}
case OP_POPSTACKDESTROY: {
if (SlowPath) {
SetX87ValidTag(0, false);
SetX87ValidTag(GetTopWithCache_Slow(), false);
}
StackPop();
break;
@@ -1145,14 +1052,12 @@ void X87StackOptimization::Run(IREmitter* Emit) {
case OP_SYNCSTACKTOSLOW: {
// This synchronizes stack values but doesn't necessarily moves us off the FastPath!
Ref NewTop = SynchronizeStackValues();
FlushCachedRegs();
IREmit->ReplaceUsesWithAfter(CodeNode, NewTop, CodeNode);
break;
}
case OP_STACKFORCESLOW: {
MigrateToSlowPathIf(true);
InvalidateCachedRegs();
break;
}
@@ -1179,7 +1084,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
Ref Value {};
if (ReducedPrecisionMode) {
Value = IREmit->_Vector_FToI(OpSize::i64Bit, OpSize::i64Bit, St0, RoundMode::Host);
Value = IREmit->_Vector_FToI(OpSize::i64Bit, OpSize::i64Bit, St0, Round_Host);
} else {
Value = IREmit->_F80Round(St0);
}
@@ -1211,7 +1116,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
LOGMAN_THROW_A_FMT(IsBlockExit(LastIROp->Op), "must be exit");
IREmit->SetWriteCursorBefore(LastCodeNode);
SynchronizeStackValues();
FlushCachedRegs();
}
return;
@@ -19,9 +19,9 @@ union PhysicalRegister {
return Raw == Other.Raw;
}
PhysicalRegister(RegClass Class, uint8_t Reg)
PhysicalRegister(RegisterClassType Class, uint8_t Reg)
: Reg(Reg)
, Class(uint8_t(Class)) {}
, Class(Class.Val) {}
PhysicalRegister(OrderedNodeWrapper Arg)
: Raw(Arg.GetImmediate()) {}
@@ -29,16 +29,12 @@ union PhysicalRegister {
PhysicalRegister(Ref Node)
: Raw(Node->Reg) {}
RegClass AsRegClass() const {
return RegClass {Class};
}
static const PhysicalRegister Invalid() {
return PhysicalRegister(RegClass::Invalid, 0);
return PhysicalRegister(InvalidClass, 0);
}
bool IsInvalid() const {
static_assert(uint8_t(RegClass::Invalid) == 0);
static_assert(InvalidClass == 0);
return Raw == 0;
}
};
-5
View File
@@ -4,7 +4,6 @@
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/PrctlUtils.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/memory.h>
@@ -51,10 +50,6 @@ void* FEX_mmap(void* addr, size_t length, int prot, int flags, int fd, off_t off
errno = -(uint64_t)Result;
return (void*)-1;
}
if (flags & MAP_ANONYMOUS) {
prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Result, length, "FEXMem");
}
return Result;
}
int FEX_munmap(void* addr, size_t length) {
@@ -165,8 +165,7 @@ private:
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
auto Res = mprotect(reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData, PROT_READ | PROT_WRITE);
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno,
strerror(errno));
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno, strerror(errno));
LiveVMARegion* LiveRange = new (reinterpret_cast<void*>(ReservedRegion->Base)) LiveVMARegion();
@@ -274,8 +273,8 @@ void* OSAllocator_64Bit::Mmap(void* addr, size_t length, int prot, int flags, in
again:
struct RangeResult final {
LiveVMARegion* RegionInsertedInto;
void* Ptr;
LiveVMARegion *RegionInsertedInto;
void *Ptr;
};
auto CheckIfRangeFits = [&AllocatedOffset](LiveVMARegion* Region, uint64_t length, int prot, int flags, int fd, off_t offset,
+8 -3
View File
@@ -78,9 +78,14 @@ When generating IR inside of the `OpDispatchBuilder` it is straight forward, jus
This is an intrusive allocator that is used by the `OpDispatchBuilder` for storing IR data. It is a simple linear arena allocator without resizing capabilities.
### OpDispatchBuilder
OpDispatchBuilder provides `IRListView ViewIR()` for handling the IR outside of the class:
* Returns a wrapper container class the allows you to view the IR. This doesn't take ownership of the IR data.
* If the OpDispatcherBuilder changes its IR then changes are also visible to this class
OpDispatchBuilder provides two routines for handling the IR outside of the class
* `IRListView ViewIR();`
* Returns a wrapper container class the allows you to view the IR. This doesn't take ownership of the IR data.
* If the OpDispatcherBuilder changes its IR then changes are also visible to this class
* `IRListView *CreateIRCopy()`
* As the name says, it creates a new copy of the IR that is in the OpDispatchBuilder
* Copying the IR only copies the memory used and doesn't have any free space for optimizations after this copy operation
* Useful for tiered recompilers, AOT, and offline analysis
This class uses two IntrusiveAllocator objects for tracking IR data. `ListData` and `Data` are the object names.
* `ListData` is for tracking the doubly linked list of nodes
-58
View File
@@ -1,58 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <cstdint>
namespace FEXCore {
namespace Core {
struct InternalThreadState;
} // namespace Core
namespace HLE {
struct SourcecodeMap;
} // namespace HLE
// Generic information associated with an executable file.
struct ExecutableFileInfo {
~ExecutableFileInfo();
fextl::unique_ptr<HLE::SourcecodeMap> SourcecodeMap;
fextl::string FileId;
fextl::string Filename;
};
// Information associated with a specific section of an executable file
struct ExecutableFileSectionInfo {
ExecutableFileInfo& FileInfo;
// Start address that the file is mapped to.
uintptr_t FileStartVA;
};
class AbstractCodeCache {
public:
virtual ~AbstractCodeCache() = default;
/**
* Loads a code cache from mapped memory and appends it to the current Core state.
* TODO: Optionally recompiles all contained code blocks at runtime for validation.
*/
virtual void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) = 0;
/**
* Bundles the current Core state (CodeBuffer, GuestToHostMapping, ...) to a code cache and writes it to the given file descriptor.
* Returns true on success.
*/
virtual bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) = 0;
/**
* Function to be called before compiling any code for caching purposes
*/
virtual void InitiateCacheGeneration() = 0;
};
} // namespace FEXCore
+17 -2
View File
@@ -4,7 +4,6 @@
#include <stdint.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/CodeCache.h>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
@@ -14,7 +13,12 @@
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <istream>
#include <ostream>
#include <span>
namespace FEXCore {
class CodeLoader;
struct HostFeatures;
class ForkableSharedMutex;
class ThunkHandler;
@@ -25,11 +29,17 @@ struct CPUState;
struct InternalThreadState;
} // namespace FEXCore::Core
namespace FEXCore::CPU {
class CPUBackend;
}
namespace FEXCore::HLE {
struct SyscallArguments;
class SyscallHandler;
} // namespace FEXCore::HLE
namespace FEXCore::IR {
struct AOTIRCacheEntry;
class IREmitter;
} // namespace FEXCore::IR
@@ -138,13 +148,18 @@ public:
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) = 0;
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) = 0;
virtual AbstractCodeCache& GetCodeCache() = 0;
FEX_DEFAULT_VISIBILITY virtual FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) = 0;
FEX_DEFAULT_VISIBILITY virtual void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) = 0;
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) = 0;
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(
FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start, uint64_t Length) = 0;
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
FEX_DEFAULT_VISIBILITY virtual void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) = 0;
FEX_DEFAULT_VISIBILITY virtual void
ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) = 0;
+6 -6
View File
@@ -146,15 +146,15 @@ struct CPUState {
// - Three are reserved for user-space to setup TLS segments in
// LDT segments are entirely controlled by userspace.
// - Kernel allocates up to 8192 ldt segments.
gdt_segment* segment_arrays[2] {};
gdt_segment *segment_arrays[2] {};
static gdt_segment* GetSegmentFromIndex(CPUState& State, uint16_t Selector) {
static gdt_segment* GetSegmentFromIndex(CPUState &State, uint16_t Selector) {
auto base = State.segment_arrays[(Selector >> 2) & 1];
return &base[Selector >> 3];
}
static uint32_t CalculateGDTBase(gdt_segment GDT) {
uint32_t Base {};
uint32_t Base{};
Base |= GDT.Base2 << 24;
Base |= GDT.Base1 << 16;
Base |= GDT.Base0;
@@ -162,19 +162,19 @@ struct CPUState {
}
static uint32_t CalculateGDTLimit(gdt_segment GDT) {
uint32_t Limit {};
uint32_t Limit{};
Limit |= GDT.Limit1 << 16;
Limit |= GDT.Limit0;
return Limit;
}
static void SetGDTBase(gdt_segment* GDT, uint32_t Base) {
static void SetGDTBase(gdt_segment *GDT, uint32_t Base) {
GDT->Base0 = Base;
GDT->Base1 = Base >> 16;
GDT->Base2 = Base >> 24;
}
static void SetGDTLimit(gdt_segment* GDT, uint32_t Limit) {
static void SetGDTLimit(gdt_segment *GDT, uint32_t Limit) {
GDT->Limit0 = Limit;
GDT->Limit1 = Limit >> 16;
}
@@ -40,7 +40,6 @@ struct HostFeatures {
bool SupportsECV {};
bool SupportsWFXT {};
bool Supports3DNow {};
bool SupportsSSE4a {};
// Float exception behaviour
bool SupportsAFP {};
@@ -9,6 +9,10 @@
#include <memory>
#include <filesystem>
namespace FEXCore::IR {
struct AOTIRCacheEntry;
}
namespace FEXCore::HLE {
struct SourcecodeLineMapping {
@@ -84,6 +88,6 @@ private:
class SourcecodeResolver {
public:
virtual fextl::unique_ptr<SourcecodeMap> GenerateMap(std::string_view GuestBinaryFile, std::string_view GuestBinaryFileId) = 0;
virtual fextl::unique_ptr<SourcecodeMap> GenerateMap(const std::string_view& GuestBinaryFile, const std::string_view& GuestBinaryFileId) = 0;
};
} // namespace FEXCore::HLE
+19 -4
View File
@@ -1,11 +1,13 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <cstdint>
#include <optional>
#include <shared_mutex>
#include <FEXCore/Core/CodeCache.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/fextl/string.h>
namespace FEXCore::IR {
struct AOTIRCacheEntry;
}
namespace FEXCore::Context {
class Context;
@@ -49,6 +51,19 @@ struct ExecutableRangeInfo {
class SyscallHandler;
class SourcecodeResolver;
struct AOTIRCacheEntryLookupResult {
AOTIRCacheEntryLookupResult(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart)
: Entry(Entry)
, VAFileStart(VAFileStart) {}
AOTIRCacheEntryLookupResult(AOTIRCacheEntryLookupResult&&) = default;
FEXCore::IR::AOTIRCacheEntry* Entry;
uintptr_t VAFileStart;
friend class SyscallHandler;
};
class SyscallHandler {
public:
virtual ~SyscallHandler() = default;
@@ -67,7 +82,7 @@ public:
virtual void MarkOvercommitRange(uint64_t Start, uint64_t Length) {}
virtual void UnmarkOvercommitRange(uint64_t Start, uint64_t Length) {}
virtual ExecutableRangeInfo QueryGuestExecutableRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) = 0;
virtual std::optional<ExecutableFileSectionInfo> LookupExecutableFileSection(Core::InternalThreadState& Thread, uint64_t GuestAddr) = 0;
virtual AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddr) = 0;
virtual void PreCompile() {}
Loaded 100 of 296 files, more files were not shown because too many files have changed in this diff. Show more