Compare commits

..
Author SHA1 Message Date
Ryan Houdek c330da4992 FEXServerClient: Workaround sun_path 108 byte limit
We really don't want to do this, but in the case that the AF_UNIX path
is longer than the 108-byte limit that sun_path provides we don't really
have a choice. The alternative choice would be to switch /entirely/ away
from AF_UNIX and instead use pipes. We need a bandage fix for now, so
throw the socket in to a temp folder if the path is too long.
2026-07-02 16:49:28 -07:00
728 changed files with 25341 additions and 102631 deletions

No files matched your search

+1 -1
View File
@@ -43,7 +43,7 @@ jobs:
distrobox upgrade steamrt4
distrobox enter --name steamrt4 -- sudo apt-get install -y \
git cmake ninja-build ccache \
lld clang clang-tools \
lld clang \
libclang-dev llvm-dev \
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross \
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
+1 -15
View File
@@ -24,7 +24,7 @@ runs:
cmake -S . -B build_${{ inputs.target }} -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=Data/CMake/toolchain_mingw.cmake \
-DMINGW_TRIPLE=${_cc}-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja \
-DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False \
-DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=generic -DTUNE_CPU=none -DRANGES_NATIVE=OFF
-DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=generic -DTUNE_CPU=none
- name: Build
shell: bash
@@ -33,17 +33,3 @@ runs:
- name: Install
shell: bash
run: DESTDIR="$PWD"/install cmake --build build_${{ inputs.target }} -t install
- name: Configure UnixLib
shell: bash
run: |
cmake -S Source/Windows/UnixLib -B build_unixlib_${{ inputs.target }} -DCMAKE_BUILD_TYPE=$BUILD_TYPE \
-G Ninja -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-unix -DCMAKE_INSTALL_PREFIX=/usr
- name: Build UnixLib
shell: bash
run: cmake --build build_unixlib_${{ inputs.target }}
- name: Install UnixLib
shell: bash
run: DESTDIR="$PWD"/install cmake --build build_unixlib_${{ inputs.target }} -t install
+1 -3
View File
@@ -50,8 +50,6 @@ jobs:
with:
overwrite: true
name: wine_dll_artifacts
path: |
${{ github.workspace }}/install/usr/lib/wine/aarch64-windows/lib*.dll
${{ github.workspace }}/install/usr/lib/wine/aarch64-unix/lib*.so
path: ${{ github.workspace }}/install/usr/lib/wine/aarch64-windows/lib*.dll
retention-days: 60
compression-level: 9
+3 -3
View File
@@ -31,12 +31,12 @@ build:
- apt-get -y update
- apt-get install -y
git cmake ninja-build ccache
lld clang clang-tools
lld clang
libclang-dev llvm-dev
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
- cmake -E make_directory build/
- cmake -DCMAKE_BUILD_TYPE=Release -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=armv8.2-a -DTUNE_CPU=none -DRANGES_NATIVE=OFF . -B build/
- cmake -DCMAKE_BUILD_TYPE=Release -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=armv8.2-a -DTUNE_CPU=none . -B build/
- cmake --build build/ --config Release
- DESTDIR=$(pwd)/install/ cmake --build build/ --config Release -t install
@@ -57,7 +57,7 @@ promote:
- arm64
- aarch64
rules:
- if: $PROMOTE_BRANCH && $CI_COMMIT_BRANCH == 'main'
- if: '$PROMOTE_BRANCH'
before_script:
- apt-get -y update
- apt-get install -y tmux curl
-1
View File
@@ -1 +0,0 @@
AI must not be used to generate code for contributions to this project.
-1
View File
@@ -1 +0,0 @@
AI must not be used to generate code for contributions to this project.
+13 -50
View File
@@ -195,9 +195,6 @@ if (ENABLE_GDB_SYMBOLS)
endif()
add_compile_definitions(_LARGEFILE64_SOURCE)
if (WIN32)
add_compile_definitions(UNICODE _UNICODE)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
@@ -256,15 +253,8 @@ endif()
if (ENABLE_CCACHE)
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
execute_process(COMMAND "${CCACHE_PROGRAM}" --print-version
OUTPUT_VARIABLE CCACHE_VERSION OUTPUT_STRIP_TRAILING_WHITESPACE)
message(STATUS "Enabling ccache ${CCACHE_VERSION}")
if (CCACHE_VERSION VERSION_GREATER_EQUAL "4.8")
# Set sloppiness to enable caching even for files that use __DATE__/__TIME__ macros
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM} sloppiness=time_macros")
else()
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
endif()
message(STATUS "CCache enabled")
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
endif()
endif()
@@ -367,10 +357,7 @@ include(LinkerGC)
## Externals ##
if (NOT BUILD_STEAM_SUPPORT)
find_package(unordered_dense QUIET CONFIG)
endif()
find_package(unordered_dense QUIET CONFIG)
if (NOT unordered_dense_FOUND)
add_subdirectory(External/unordered_dense)
endif()
@@ -381,10 +368,8 @@ if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
endif()
if (ENABLE_ZYDIS)
if (NOT BUILD_STEAM_SUPPORT)
find_package(Zycore 1.5 MODULE QUIET)
find_package(Zydis 4.0 MODULE QUIET)
endif()
find_package(Zycore 1.5 MODULE QUIET)
find_package(Zydis 4.0 MODULE QUIET)
if (TARGET Zydis::Zydis AND TARGET Zycore::Zycore)
message(STATUS "Using system Zydis")
@@ -405,7 +390,7 @@ find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
set(BUILD_SHARED_LIBS OFF)
if (NOT CMAKE_CROSSCOMPILING AND NOT BUILD_STEAM_SUPPORT)
if (NOT CMAKE_CROSSCOMPILING)
find_package(xxhash MODULE QUIET)
endif()
@@ -433,22 +418,14 @@ else ()
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
endif()
if (MINGW OR BUILD_STEAM_SUPPORT)
find_package(fmt QUIET)
if (NOT fmt_FOUND)
# Disable fmt install
set(FMT_INSTALL OFF)
add_subdirectory(External/fmt/)
else()
find_package(fmt QUIET)
if (NOT fmt_FOUND)
# Disable fmt install
set(FMT_INSTALL OFF)
add_subdirectory(External/fmt/)
endif()
endif()
if (NOT BUILD_STEAM_SUPPORT)
find_package(range-v3 QUIET)
endif()
find_package(range-v3 QUIET)
if (NOT range-v3_FOUND)
add_subdirectory(External/range-v3/)
target_compile_definitions(range-v3 INTERFACE RANGES_DISABLE_DEPRECATED_WARNINGS)
@@ -493,20 +470,12 @@ endif()
set(FEX_TUNE_COMPILE_FLAGS)
if (NOT TUNE_ARCH STREQUAL "generic")
set(TUNE_ARCH_STRING "${TUNE_ARCH}")
if(ARCHITECTURE_arm64)
set(TUNE_ARCH_STRING "${TUNE_ARCH}+crc")
endif()
check_cxx_compiler_flag("-march=${TUNE_ARCH_STRING}" COMPILER_SUPPORTS_ARCH_TYPE)
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
if(COMPILER_SUPPORTS_ARCH_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH_STRING}")
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH}")
else()
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH_STRING}' but the compiler doesn't support this")
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
endif()
elseif(ARCHITECTURE_arm64)
# Need to always append crc
check_cxx_compiler_flag("-march=armv8-a+crc" COMPILER_SUPPORTS_ARCH_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=armv8-a+crc")
endif()
if (TUNE_CPU STREQUAL "native")
@@ -617,12 +586,6 @@ if (BUILD_TESTING)
execute_process(COMMAND "nproc" OUTPUT_STRIP_TRAILING_WHITESPACE OUTPUT_VARIABLE TEST_JOB_COUNT)
endif()
set(TEST_JOB_FLAG "-j${TEST_JOB_COUNT}")
# Runs the whole test suite, with large pages enabled to reduce the cost of fork() in ctest.
add_custom_target(tests
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}"
USES_TERMINAL
COMMAND ${CMAKE_COMMAND} -E env GLIBC_TUNABLES=glibc.malloc.hugetlb=1 ctest "--progress" "--timeout" "302" ${TEST_JOB_FLAG})
endif()
add_subdirectory(External/SoftFloat-3e/)
+132
View File
@@ -0,0 +1,132 @@
{
"environments": [
{
"BuildPath": "${projectDir}\\out\\build\\${name}",
"InstallPath": "${projectDir}\\out\\install\\${name}",
"clangcl": "clang-cl.exe",
"cc": "clang",
"cxx": "clang++"
}
],
"configurations": [
{
"name": "WSL-Clang-Debug",
"generator": "Ninja",
"configurationType": "Debug",
"buildRoot": "${env.BuildPath}",
"installRoot": "${env.InstallPath}",
"cmakeExecutable": "/usr/bin/cmake",
"cmakeCommandArgs": "",
"buildCommandArgs": "-v",
"ctestCommandArgs": "",
"wslPath": "${defaultWSLPath}",
"inheritEnvironments": [ "linux_clang_x64" ],
"addressSanitizerRuntimeFlags": "detect_leaks=0",
"variables": [
{
"name": "WSL",
"value": "TRUE",
"type": "BOOL"
}
]
},
{
"name": "WSL-Clang-Release",
"generator": "Ninja",
"configurationType": "RelWithDebInfo",
"buildRoot": "${env.BuildPath}",
"installRoot": "${env.InstallPath}",
"cmakeExecutable": "/usr/bin/cmake",
"cmakeCommandArgs": "",
"buildCommandArgs": "-v",
"ctestCommandArgs": "",
"wslPath": "${defaultWSLPath}",
"inheritEnvironments": [ "linux_clang_x64" ],
"addressSanitizerRuntimeFlags": "detect_leaks=0",
"variables": [
{
"name": "WSL",
"value": "TRUE",
"type": "BOOL"
}
]
},
{
"name": "x86-Clang-Cross-Debug",
"generator": "Ninja",
"configurationType": "Debug",
"buildRoot": "${env.BuildPath}",
"installRoot": "${env.InstallPath}",
"cmakeCommandArgs": "",
"buildCommandArgs": "-v",
"ctestCommandArgs": "",
"inheritEnvironments": [ "clang_cl_x86" ],
"variables": [
{
"name": "CMAKE_C_COMPILER",
"value": "${env.cc}",
"type": "STRING"
},
{
"name": "CMAKE_CXX_COMPILER",
"value": "${env.cxx}",
"type": "STRING"
},
{
"name": "CMAKE_SYSROOT",
"value": "${env.fexsysroot}",
"type": "STRING"
}
]
},
{
"name": "x64-Clang-Cross-Release",
"generator": "Ninja",
"configurationType": "RelWithDebInfo",
"buildRoot": "${env.BuildPath}",
"installRoot": "${env.InstallPath}",
"cmakeCommandArgs": "",
"buildCommandArgs": "-v",
"ctestCommandArgs": "",
"inheritEnvironments": [ "clang_cl_x86" ],
"variables": [
{
"name": "CMAKE_C_COMPILER",
"value": "${env.cc}",
"type": "STRING"
},
{
"name": "CMAKE_CXX_COMPILER",
"value": "${env.cxx}",
"type": "STRING"
},
{
"name": "CMAKE_SYSROOT",
"value": "${env.fexsysroot}",
"type": "STRING"
}
]
},
{
"name": "Linux-Clang-Remote-Debug",
"generator": "Ninja",
"configurationType": "Debug",
"cmakeExecutable": "/usr/bin/cmake",
"remoteCopySourcesExclusionList": [ ".vs", ".vscode", ".git", ".github", "build", "out", "bin" ],
"cmakeCommandArgs": "",
"buildCommandArgs": "-v",
"ctestCommandArgs": "",
"inheritEnvironments": [ "linux_clang_x64" ],
"remoteMachineName": "${env.fexremote}",
"remoteCMakeListsRoot": "$HOME/projects/.vs/${projectDirName}/src",
"remoteBuildRoot": "$HOME/projects/.vs/${projectDirName}/build/${name}",
"remoteInstallRoot": "$HOME/projects/.vs/${projectDirName}/install/${name}",
"remoteCopySources": true,
"rsyncCommandArgs": "-t --delete --delete-excluded",
"remoteCopyBuildOutput": false,
"remoteCopySourcesMethod": "rsync",
"addressSanitizerRuntimeFlags": "detect_leaks=0",
"variables": []
}
]
}
-1
View File
@@ -1 +0,0 @@
No AI/ML/LLM/etc code contributions.
+11 -18
View File
@@ -270,9 +270,6 @@ public:
void fcvtxnt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
SVEFloatConvertOdd(0b00, 0b10, pg, zn, zd);
}
void bfcvtnt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
SVEFloatConvertOdd(0b10, 0b10, pg, zn, zd);
}
///< Size is destination size
void fcvtnt(SubRegSize size, ZRegister zd, PRegisterMerge pg, ZRegister zn) {
LOGMAN_THROW_A_FMT(size == SubRegSize::i32Bit || size == SubRegSize::i16Bit, "Unsupported size in {}", __func__);
@@ -295,6 +292,8 @@ public:
SVEFloatConvertOdd(ConvertedSrcSize, ConvertedDestSize, pg, zn, zd);
}
// XXX: BFCVTNT
// SVE2 floating-point pairwise operations
void faddp(SubRegSize size, ZRegister zd, PRegisterMerge pg, ZRegister zn, ZRegister zm) {
SVEFloatPairwiseArithmetic(0b000, size, pg, zd, zn, zm);
@@ -2313,15 +2312,15 @@ public:
// SVE floating-point convert precision
void fcvt(SubRegSize to, SubRegSize from, ZRegister zd, PRegisterMerge pg, ZRegister zn) {
LOGMAN_THROW_A_FMT(to != from, "to and from sizes cannot be the same.");
LOGMAN_THROW_A_FMT(to != SubRegSize::i8Bit && from != SubRegSize::i8Bit, "Can't use 8-bit element size");
SVEFPConvertPrecision(to, from, zd, pg, zn);
}
void fcvtx(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
SVEFPConvertPrecision(SubRegSize::i32Bit, SubRegSize::i8Bit, zd, pg, zn);
}
void bfcvt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
SVEFPConvertPrecision(SubRegSize::i32Bit, SubRegSize::i32Bit, zd, pg, zn);
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0101'0000'1010'1010'0000'0000'0000;
Instr |= pg.Idx() << 10;
Instr |= zn.Idx() << 5;
Instr |= zd.Idx();
dc32(Instr);
}
// SVE floating-point unary operations
@@ -3848,19 +3847,14 @@ private:
void SVEFPConvertPrecision(SubRegSize to, SubRegSize from, ZRegister zd, PRegister pg, ZRegister zn) {
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_A_FMT(to != SubRegSize::i128Bit && from != SubRegSize::i128Bit, "Can't use 128-bit element size");
LOGMAN_THROW_A_FMT(to != from, "to and from sizes cannot be the same.");
LOGMAN_THROW_A_FMT(to != SubRegSize::i8Bit && to != SubRegSize::i128Bit && from != SubRegSize::i8Bit && from != SubRegSize::i128Bit,
"Can't use 8-bit or 128-bit element size");
// Encodings for the to and from sizes can get a little funky
// depending on what is being converted to/from.
const uint32_t op = [&] {
switch (from) {
case SubRegSize::i8Bit: {
switch (to) {
case SubRegSize::i32Bit: return 0x00020000U;
default: return UINT32_MAX;
}
}
case SubRegSize::i16Bit: {
switch (to) {
case SubRegSize::i32Bit: return 0x00810000U;
@@ -3872,7 +3866,6 @@ private:
case SubRegSize::i32Bit: {
switch (to) {
case SubRegSize::i16Bit: return 0x00800000U;
case SubRegSize::i32Bit: return 0x00820000U;
case SubRegSize::i64Bit: return 0x00C30000U;
default: return UINT32_MAX;
}
+2 -2
View File
@@ -9,8 +9,8 @@ set(CMAKE_AR ${MINGW_TRIPLE}-ar)
# Compile everything as static to avoid requiring the MinGW runtime libraries, force page aligned sections so that
# debug symbols work correctly, and disable loop alignment to workaround an LLVM bug
# (https://github.com/llvm/llvm-project/issues/47432)
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-static -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-static -static-libgcc -static-libstdc++ -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -static-libgcc -static-libstdc++ -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
set(CMAKE_C_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
set(CMAKE_CXX_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
set(CMAKE_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
-7
View File
@@ -46,13 +46,6 @@
"@PREFIX_LIB@/libwayland-client.so.0",
"@PREFIX_LIB@/libwayland-client.so.0.20.0"
]
},
"cuda": {
"Library" : "libcuda-guest.so",
"Overlay": [
"@PREFIX_LIB@/libcuda.so",
"@PREFIX_LIB@/libcuda.so.1"
]
}
}
}
+63 -60
View File
@@ -1,5 +1,5 @@
#
# This file is autogenerated by pip-compile with Python 3.14
# This file is autogenerated by pip-compile with Python 3.13
# by the following command:
#
# pip-compile --generate-hashes --output-file=requirements_formatting.txt --strip-extras requirements_formatting.txt.in
@@ -210,53 +210,56 @@ click==8.1.7 \
--hash=sha256:ae74fb96c20a0277a1d615f1e4d73c8414f5a98db8b799a7931d1582f3390c28 \
--hash=sha256:ca9853ad459e787e2192211578cc907e7594e294c7ccc834310722b41b9ca6de
# via black
cryptography==50.0.0 \
--hash=sha256:031e2d5dd4bb9caa3ca9c82e5a197fd8ae680232cee62603d1a813f3f07e3d03 \
--hash=sha256:06a32a980526a6ab9a4b9bf8f7385800791e2bb960903cb6b530e4817509a3b7 \
--hash=sha256:07479a1cb08219ab719147e742e76090c9c773321959bb94946fffdd397a6437 \
--hash=sha256:07949c449a1abcf60d1ee6e88956d89404c7df3c8258f46589e912988e551987 \
--hash=sha256:105110f43a471dbd0060b9c9516cb8a6a79233631a04cc2ba16f28323ac6e025 \
--hash=sha256:11b74db56cdbe3cdee6e3f6982ecb70334fa10dce99ed58bf7894aaaa3b2a037 \
--hash=sha256:12b9c6996425c76ea6c457ace4f3073e715b8c545add07cd1a8f3a4f90691269 \
--hash=sha256:1489e263a8048bb8b6a8bac662eb2d402ea5d2b7b4699b72f385f1e2772db105 \
--hash=sha256:19736989797678c6af1e55cd49055cdbcb55d8f6b5583ac5335f933aba9101dc \
--hash=sha256:1b4a266766514614f8aa60416e71f2fc6e575d36e7bdc90f644fadb2f4b75b95 \
--hash=sha256:2a8183b489dc1f7f80f135780fadc1108f14b31b8a40411c7a5b17425f65f28b \
--hash=sha256:37fdb0d0111f1e2ff07139dfb79f1b49531f8e213c46f1163dd7642979b58c47 \
--hash=sha256:3f5735ffe4996d28b809371756219f5354864902a3b9e7c0b9ee87041209fc9c \
--hash=sha256:49e7d93abdbd2990caced757e5fade25302f719c3c8fb6e6fff2dde98999fc41 \
--hash=sha256:5e34edd123674534acd70147f0ca331eaa2c74e6325fb2028c886aa26ba0b68c \
--hash=sha256:62598a8a57f815db4c6259a4e97d857dab56697e7de8e8ab02352ab74da1995d \
--hash=sha256:65c2c3add92b45fd0709db8594536aea39c2a67af0e27ffcf049c498501140b7 \
--hash=sha256:6ba6a53445bd3cfa809ef3ef5f1589aa6ba08784a1d962bf47d0940e871dab1c \
--hash=sha256:6e7d61120573a7f2cd94cc095f9e81f6967c61ccdf194285aa143ecec8e0b708 \
--hash=sha256:7cec5b856506da6defb290f30c9ee687d5f5e8cb0bd3f6459dde43b0b4fa40ef \
--hash=sha256:80b63928fa35083b33966ce1efb70e5b9607181e49dcd1c22c8c005e319f667f \
--hash=sha256:82148ec5bddac30b51a5b3c1945075f896fa022cb93f8e4a01e9f6ee95292c5f \
--hash=sha256:828743d939e9629bc267b8e2d08d8bb67cd4319c771a33d4b18b22dd8fb7440a \
--hash=sha256:8d89f3976b10b4ce31118de72329025f70d2c6ead14a8217c5514dd2c6d5a78f \
--hash=sha256:8eb5e1172eb569ea8a872796576e6a67c276351728b6455d5beb01242b027c6a \
--hash=sha256:900131fafd8aead39ac7dd3a7e833be754c17a95cfd91221636949fe4eb0aa8a \
--hash=sha256:910d11e1a385c654bf738bf3e6b8e6ed5de0f5610fcae2be9e5b398d8081d20e \
--hash=sha256:910e1d2668e7de9648f2bcee30e180db2a6b15c30f887d7c4c93ddf96e3992e3 \
--hash=sha256:9aa87839c383bdbab6ef865787a1fb877af8dd03464c4400322726feaaadfc6d \
--hash=sha256:a1b30560f2acc95aa8b2e06e716a13dbfc97314747b80d9707e307f77b40d6b3 \
--hash=sha256:a91296cb61e8df6f86d0c19cc4068228da256bf59bf86049fbd821084565327f \
--hash=sha256:b42a28c1844fd9de8f3f7d540e36b66f3a9c83fceac7170ebc7a6a19edd9dcae \
--hash=sha256:bd1c592e4d5974f0d08d4888e432157adba757c66da0246918e43677fafa2d30 \
--hash=sha256:c87f62a3d3b9888ed0fdde100ec06aa61ca9cd44bad9057d1dff9a516b5f5bb9 \
--hash=sha256:c99c003e088647b8a5b7c145d6f78c335f6348332b62e142d411c4b63d1460b9 \
--hash=sha256:ccdc4a71a4dabae05de219404f9f4abc38e3b58422177ff93d0da05967dafa07 \
--hash=sha256:d24fead1d4d076e1bfb006dcec392074a3cd8d7b4fc8a595aa64073b2b7a96ba \
--hash=sha256:d58c3db7cd6eed54e6c06744db55456b65ebd7492ddeae9c1e93cfca7aa857d3 \
--hash=sha256:d764dcf130c428ef66786f866dd750f53182bc608813489915e9fc106bb0c82f \
--hash=sha256:df2a58a472f332225671c35b0a830208b86d004f82baa8530fa3782c85646533 \
--hash=sha256:e722f16708d854fe924790e051061f6704a472c3bac347b6fd88033ea8dd0dc5 \
--hash=sha256:ecfed7367f965a0328cfbdd70da860f15441f002f613185668c6e6ebf5a0ac11 \
--hash=sha256:eeac2acb5a20ed25e0ad6d1df9891a520b78b404266b6d11778f25d5d691a6c9 \
--hash=sha256:f59e38625469987d7ef6d495323c55e7db6c212eaf6112267e0d3b565a2e9c9f \
--hash=sha256:f89831ef99dd7dd169ab06d63a831adb9e20a87aac6d380266bbda5823349169 \
--hash=sha256:fd9192b7b70c573d7f214eb1ae35e00d359f6f5e4b27c7e21e30de1fc6204645
cryptography==46.0.5 \
--hash=sha256:02f547fce831f5096c9a567fd41bc12ca8f11df260959ecc7c3202555cc47a72 \
--hash=sha256:039917b0dc418bb9f6edce8a906572d69e74bd330b0b3fea4f79dab7f8ddd235 \
--hash=sha256:1abfdb89b41c3be0365328a410baa9df3ff8a9110fb75e7b52e66803ddabc9a9 \
--hash=sha256:2ae6971afd6246710480e3f15824ed3029a60fc16991db250034efd0b9fb4356 \
--hash=sha256:2b7a67c9cd56372f3249b39699f2ad479f6991e62ea15800973b956f4b73e257 \
--hash=sha256:351695ada9ea9618b3500b490ad54c739860883df6c1f555e088eaf25b1bbaad \
--hash=sha256:38946c54b16c885c72c4f59846be9743d699eee2b69b6988e0a00a01f46a61a4 \
--hash=sha256:3b4995dc971c9fb83c25aa44cf45f02ba86f71ee600d81091c2f0cbae116b06c \
--hash=sha256:3ce58ba46e1bc2aac4f7d9290223cead56743fa6ab94a5d53292ffaac6a91614 \
--hash=sha256:3ee190460e2fbe447175cda91b88b84ae8322a104fc27766ad09428754a618ed \
--hash=sha256:4108d4c09fbbf2789d0c926eb4152ae1760d5a2d97612b92d508d96c861e4d31 \
--hash=sha256:420d0e909050490d04359e7fdb5ed7e667ca5c3c402b809ae2563d7e66a92229 \
--hash=sha256:47fb8a66058b80e509c47118ef8a75d14c455e81ac369050f20ba0d23e77fee0 \
--hash=sha256:4c3341037c136030cb46e4b1e17b7418ea4cbd9dd207e4a6f3b2b24e0d4ac731 \
--hash=sha256:4d7e3d356b8cd4ea5aff04f129d5f66ebdc7b6f8eae802b93739ed520c47c79b \
--hash=sha256:4d8ae8659ab18c65ced284993c2265910f6c9e650189d4e3f68445ef82a810e4 \
--hash=sha256:4e817a8920bfbcff8940ecfd60f23d01836408242b30f1a708d93198393a80b4 \
--hash=sha256:50bfb6925eff619c9c023b967d5b77a54e04256c4281b0e21336a130cd7fc263 \
--hash=sha256:556e106ee01aa13484ce9b0239bca667be5004efb0aabbed28d353df86445595 \
--hash=sha256:582f5fcd2afa31622f317f80426a027f30dc792e9c80ffee87b993200ea115f1 \
--hash=sha256:5be7bf2fb40769e05739dd0046e7b26f9d4670badc7b032d6ce4db64dddc0678 \
--hash=sha256:60ee7e19e95104d4c03871d7d7dfb3d22ef8a9b9c6778c94e1c8fcc8365afd48 \
--hash=sha256:61aa400dce22cb001a98014f647dc21cda08f7915ceb95df0c9eaf84b4b6af76 \
--hash=sha256:68f68d13f2e1cb95163fa3b4db4bf9a159a418f5f6e7242564fc75fcae667fd0 \
--hash=sha256:7d1f30a86d2757199cb2d56e48cce14deddf1f9c95f1ef1b64ee91ea43fe2e18 \
--hash=sha256:7d731d4b107030987fd61a7f8ab512b25b53cef8f233a97379ede116f30eb67d \
--hash=sha256:803812e111e75d1aa73690d2facc295eaefd4439be1023fefc4995eaea2af90d \
--hash=sha256:80a8d7bfdf38f87ca30a5391c0c9ce4ed2926918e017c29ddf643d0ed2778ea1 \
--hash=sha256:8293f3dea7fc929ef7240796ba231413afa7b68ce38fd21da2995549f5961981 \
--hash=sha256:8456928655f856c6e1533ff59d5be76578a7157224dbd9ce6872f25055ab9ab7 \
--hash=sha256:890bcb4abd5a2d3f852196437129eb3667d62630333aacc13dfd470fad3aaa82 \
--hash=sha256:94a76daa32eb78d61339aff7952ea819b1734b46f73646a07decb40e5b3448e2 \
--hash=sha256:9f16fbdf4da055efb21c22d81b89f155f02ba420558db21288b3d0035bafd5f4 \
--hash=sha256:a3d1fae9863299076f05cb8a778c467578262fae09f9dc0ee9b12eb4268ce663 \
--hash=sha256:a3d507bb6a513ca96ba84443226af944b0f7f47dcc9a399d110cd6146481d24c \
--hash=sha256:abace499247268e3757271b2f1e244b36b06f8515cf27c4d49468fc9eb16e93d \
--hash=sha256:ba2a27ff02f48193fc4daeadf8ad2590516fa3d0adeeb34336b96f7fa64c1e3a \
--hash=sha256:bc84e875994c3b445871ea7181d424588171efec3e185dced958dad9e001950a \
--hash=sha256:bfd56bb4b37ed4f330b82402f6f435845a5f5648edf1ad497da51a8452d5d62d \
--hash=sha256:c18ff11e86df2e28854939acde2d003f7984f721eba450b56a200ad90eeb0e6b \
--hash=sha256:c3bcce8521d785d510b2aad26ae2c966092b7daa8f45dd8f44734a104dc0bc1a \
--hash=sha256:c4143987a42a2397f2fc3b4d7e3a7d313fbe684f67ff443999e803dd75a76826 \
--hash=sha256:c69fd885df7d089548a42d5ec05be26050ebcd2283d89b3d30676eb32ff87dee \
--hash=sha256:ced80795227d70549a411a4ab66e8ce307899fad2220ce5ab2f296e687eacde9 \
--hash=sha256:d66e421495fdb797610a08f43b05269e0a5ea7f5e652a89bfd5a7d3c1dee3648 \
--hash=sha256:d861ee9e76ace6cf36a6a89b959ec08e7bc2493ee39d07ffe5acb23ef46d27da \
--hash=sha256:e9251e3be159d1020c4030bd2e5f84d6a43fe54b6c19c12f51cde9542a2817b2 \
--hash=sha256:f145bba11b878005c496e93e257c1e88f154d278d2638e6450d17e0f31e558d2 \
--hash=sha256:fe346b143ff9685e40192a4960938545c699054ba11d4f9029f94751e3f71d87
# via
# -r requirements_formatting.txt.in
# pyjwt
@@ -278,9 +281,9 @@ graylint==1.1.1 \
--hash=sha256:0fd8e02972ca03d0ef2bf0adea76b5343efcd492d7afb5f658f3e3a724f55a36 \
--hash=sha256:b7e0eab6c159684dbf5ef84e942c3340f6a6549b02a3d11b1a1763cc4f8f0593
# via darker
idna==3.16 \
--hash=sha256:cc246e3a3f89580c3a951b5ad298ca4638078b2cdd4f115654332b5c26daded5 \
--hash=sha256:d7a6da03db833450fca25d2358ac9ff06cd624577a4aea3a596d5c0f77b8e03d
idna==3.10 \
--hash=sha256:12f65c9b470abda6dc35cf8e63cc574b1c52b11df2c86030af0ac09b01b13ea9 \
--hash=sha256:946d195a0d259cbba61165e88e65941f16e9b36ea6ddb97f00452bae8b1287d3
# via
# -r requirements_formatting.txt.in
# requests
@@ -308,9 +311,9 @@ pygithub==2.6.1 \
--hash=sha256:6f2fa6d076ccae475f9fc392cc6cdbd54db985d4f69b8833a28397de75ed6ca3 \
--hash=sha256:b5c035392991cca63959e9453286b41b54d83bf2de2daa7d7ff7e4312cebf3bf
# via -r requirements_formatting.txt.in
pyjwt==2.15.1 \
--hash=sha256:42d59d631f7768a1028a64c7ff581a9bf7519804daf91fc5b6c56e30eec5e193 \
--hash=sha256:4f259e80cdfb6b3fc18a7de51fd1ef9ec79652f25019bae68975ca2468a34df8
pyjwt==2.12.1 \
--hash=sha256:28ca37c070cad8ba8cd9790cd940535d40274d22f80ab87f3ac6a713e6e8454c \
--hash=sha256:c74a7a2adf861c04d002db713dd85f84beb242228e671280bf709d765b03672b
# via
# -r requirements_formatting.txt.in
# pygithub
@@ -387,9 +390,9 @@ pytokens==0.4.1 \
--hash=sha256:ee44d0f85b803321710f9239f335aafe16553b39106384cef8e6de40cb4ef2f6 \
--hash=sha256:f66a6bbe741bd431f6d741e617e0f39ec7257ca1f89089593479347cc4d13324
# via black
requests==2.34.2 \
--hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 \
--hash=sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed
requests==2.32.4 \
--hash=sha256:27babd3cda2a6d50b30443204ee89830707d396671944c998b5975b031ac2b2c \
--hash=sha256:27d0316682c8a29834d3264820024b62a36942083d52caf2f14c0591336d3422
# via
# -r requirements_formatting.txt.in
# pygithub
@@ -403,9 +406,9 @@ typing-extensions==4.14.1 \
--hash=sha256:38b39f4aeeab64884ce9f74c94263ef78f3c22467c8724005483154c26648d36 \
--hash=sha256:d1e1e3b58374dc93031d6eda2420a48ea44a36c2b4766a4fdeb3710755731d76
# via pygithub
urllib3==2.7.0 \
--hash=sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c \
--hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897
urllib3==2.6.3 \
--hash=sha256:1b62b6884944a57dbe321509ab94fd4d3b307075e0c2eae991ac71ee15ad38ed \
--hash=sha256:bf272323e553dfb2e87d9bfd225ca7b0f467b919d7bbd355436d3fd37cb0acd4
# via
# -r requirements_formatting.txt.in
# pygithub
+5 -5
View File
@@ -1,10 +1,10 @@
black>=26.3.1
darker==2.1.1
PyGithub==2.6.1
cryptography>=50.0.0
urllib3>=2.7.0
requests>=2.33.0
idna>=3.15
cryptography>=46.0.5
urllib3>=2.6.3
requests>=2.32.4
idna>=3.7
certifi>=2024.7.4
PyNaCl>=1.6.2
PyJWT>=2.15.1
PyJWT>=2.12.1
+1 -1
+1 -1
+2 -2
View File
@@ -8,8 +8,8 @@ This project aims to provide a fast and functional x86-64 emulation library that
* Support a tiered recompiler to allow for fast runtime performance
* Support offline compilation and offline tooling for inspection and performance analysis
* Support threaded emulation. Including emulating x86-64's strong memory model on weak memory model architectures
* Support a majority of the x86-64 instruction space.
* Including MMX, SSE, SSE2, SSE3, SSSE3, SSE4*, AVX, AVX2, F16C, and AVX-VNNI
* Support a significant portion of the x86-64 instruction space.
* Including MMX, SSE, SSE2, SSE3, SSSE3, and SSE4*
* Support fallback routines for uncommonly used x86-64 instructions
* Including x87 and 3DNow!
* Only support userspace emulation.
-28
View File
@@ -407,32 +407,6 @@ def print_parse_enum_options(options):
output_argloader.write("#endif\n")
def print_affects_codegen_options(options, unnamed_options):
output_argloader.write("#ifdef CONFIG_AFFECTSCODEGEN\n")
output_argloader.write("#undef CONFIG_AFFECTSCODEGEN\n")
TotalConfigOptions = 0
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
TotalConfigOptions += 1
for op_group, group_vals in unnamed_options.items():
for op_key, op_vals in group_vals.items():
TotalConfigOptions += 1
output_argloader.write("constexpr static std::array<bool, {}> Config_AffectsCodeGen = {{{{\n".format(TotalConfigOptions))
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
assert "AffectsCodeGen" in op_vals, "All config options must be marked if they affect codegen."
output_argloader.write("\t{}, // {}\n".format(op_vals["AffectsCodeGen"], op_key))
for op_group, group_vals in unnamed_options.items():
for op_key, op_vals in group_vals.items():
assert "AffectsCodeGen" in op_vals, "All config options must be marked if they affect codegen."
output_argloader.write("\t{}, // {}\n".format(op_vals["AffectsCodeGen"], op_key))
output_argloader.write("}};\n")
output_argloader.write("#endif\n")
if (len(sys.argv) < 5):
sys.exit()
@@ -477,6 +451,4 @@ print_parse_jsonloader_options(options);
# Generate enum variable options
print_parse_enum_options(options);
print_affects_codegen_options(options, unnamed_options);
output_argloader.close()
+60 -64
View File
@@ -251,10 +251,6 @@ def parse_ops(ops):
if "Desc" in op_val:
OpDef.Desc = op_val["Desc"]
if not isinstance(OpDef.Desc, list):
ExitError(f"Desc field for op {OpDef.Name} must be an array of strings")
if not all(isinstance(item, str) for item in OpDef.Desc):
ExitError(f"Desc field for op {OpDef.Name} must only contain strings")
if "DynamicDispatch" in op_val:
OpDef.DynamicDispatch = bool(op_val["DynamicDispatch"])
@@ -607,77 +603,77 @@ def print_validation(op):
def print_ir_allocator_helpers():
output_file.write("#ifdef IROP_ALLOCATE_HELPERS\n")
output_file.write("\ttemplate <class T>\n"
"\tstruct Wrapper final {\n"
"\t\tT *first;\n"
"\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n"
"\n"
"\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n"
"\t\toperator OrderedNode *() { return Node; }\n"
"\t\toperator const OrderedNode *() const { return Node; }\n"
"\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n"
"\t};\n")
output_file.write("\ttemplate <class T>\n")
output_file.write("\tstruct Wrapper final {\n")
output_file.write("\t\tT *first;\n")
output_file.write("\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n")
output_file.write("\n")
output_file.write("\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n")
output_file.write("\t\toperator OrderedNode *() { return Node; }\n")
output_file.write("\t\toperator const OrderedNode *() const { return Node; }\n")
output_file.write("\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n")
output_file.write("\t};\n")
output_file.write("\ttemplate <class T>\n"
"\tusing IRPair = Wrapper<T>;\n\n")
output_file.write("\ttemplate <class T>\n")
output_file.write("\tusing IRPair = Wrapper<T>;\n\n")
output_file.write("\tIRPair<IROp_Header> AllocateRawOp(size_t HeaderSize) {\n"
"\t\tauto Op = reinterpret_cast<IROp_Header*>(DualListData.DataAllocate(HeaderSize));\n"
"\t\tmemset(Op, 0, HeaderSize);\n"
"\t\tOp->Op = IROps::OP_DUMMY;\n"
"\t\treturn IRPair<IROp_Header>{Op, CreateNode(Op)};\n"
"\t}\n\n")
output_file.write("\tIRPair<IROp_Header> AllocateRawOp(size_t HeaderSize) {\n")
output_file.write("\t\tauto Op = reinterpret_cast<IROp_Header*>(DualListData.DataAllocate(HeaderSize));\n")
output_file.write("\t\tmemset(Op, 0, HeaderSize);\n")
output_file.write("\t\tOp->Op = IROps::OP_DUMMY;\n")
output_file.write("\t\treturn IRPair<IROp_Header>{Op, CreateNode(Op)};\n")
output_file.write("\t}\n\n")
output_file.write("\ttemplate<class T, IROps T2>\n"
"\tT *AllocateOrphanOp() {\n"
"\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n"
"\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n"
"\t\tmemset(Op, 0, Size);\n"
"\t\tOp->Header.Op = T2;\n"
"\t\treturn Op;\n"
"\t}\n\n")
output_file.write("\ttemplate<class T, IROps T2>\n")
output_file.write("\tT *AllocateOrphanOp() {\n")
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
output_file.write("\t\tmemset(Op, 0, Size);\n")
output_file.write("\t\tOp->Header.Op = T2;\n")
output_file.write("\t\treturn Op;\n")
output_file.write("\t}\n\n")
output_file.write("\ttemplate<class T, IROps T2>\n"
"\tIRPair<T> AllocateOp() {\n"
"\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n"
"\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n"
"\t\tmemset(Op, 0, Size);\n"
"\t\tOp->Header.Op = T2;\n"
"\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n"
"\t}\n\n")
output_file.write("\ttemplate<class T, IROps T2>\n")
output_file.write("\tIRPair<T> AllocateOp() {\n")
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
output_file.write("\t\tmemset(Op, 0, Size);\n")
output_file.write("\t\tOp->Header.Op = T2;\n")
output_file.write("\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n")
output_file.write("\t}\n\n")
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n"
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
"\t\treturn HeaderOp->Size;\n"
"\t}\n\n")
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\t\treturn HeaderOp->Size;\n")
output_file.write("\t}\n\n")
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n"
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
"\t\treturn HeaderOp->ElementSize;\n"
"\t}\n\n")
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\t\treturn HeaderOp->ElementSize;\n")
output_file.write("\t}\n\n")
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n"
"\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n"
"\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n"
"\t}\n\n")
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n")
output_file.write("\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n")
output_file.write("\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n")
output_file.write("\t}\n\n")
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n"
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
"\t\treturn GetHasDest(HeaderOp->Op);\n"
"\t}\n\n")
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\t\treturn GetHasDest(HeaderOp->Op);\n")
output_file.write("\t}\n\n")
output_file.write("\tIROps GetOpType(const OrderedNode *Op) const {\n"
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
"\t\treturn HeaderOp->Op;\n"
"\t}\n\n")
output_file.write("\tIROps GetOpType(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\t\treturn HeaderOp->Op;\n")
output_file.write("\t}\n\n")
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n"
"\t\treturn GetRegClass(GetOpType(Op));\n"
"\t}\n\n")
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n")
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
output_file.write("\t}\n\n")
output_file.write("\tstd::string_view const& GetOpName(const OrderedNode *Op) const {\n"
"\t\treturn IR::GetName(GetOpType(Op));\n"
"\t}\n\n")
output_file.write("\tstd::string_view const& GetOpName(const OrderedNode *Op) const {\n")
output_file.write("\t\treturn IR::GetName(GetOpType(Op));\n")
output_file.write("\t}\n\n")
# Generate helpers with operands
for op in IROps:
-9
View File
@@ -4,7 +4,6 @@ set(FEXCORE_BASE_SRCS
Interface/Config/Config.cpp
Utils/Allocator.cpp
Utils/FileLoading.cpp
Utils/FileUtils.cpp
Utils/ForcedAssert.cpp
Utils/LogManager.cpp
Utils/SpinWaitLock.cpp
@@ -19,14 +18,12 @@ set(SRCS
Common/JitSymbols.cpp
Interface/Context/Context.cpp
Interface/Core/LookupCache.cpp
Interface/Core/DiskCache.cpp
Interface/Core/CodeCache.cpp
Interface/Core/Core.cpp
Interface/Core/CPUBackend.cpp
Interface/Core/Addressing.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/SharedCodeBufferManager.cpp
Interface/Core/OpcodeDispatcher/AVX_128.cpp
Interface/Core/OpcodeDispatcher/Crypto.cpp
Interface/Core/OpcodeDispatcher/Flags.cpp
@@ -71,7 +68,6 @@ set(SRCS
Utils/LongJump.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/WorkQueueThread.cpp
Utils/Profiler.cpp)
if (ARCHITECTURE_arm64)
@@ -262,10 +258,6 @@ endfunction()
# Build FEXCore_Base static library
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
target_link_libraries(FEXCore_Base PUBLIC ${LIBS})
if (MINGW)
target_link_libraries(FEXCore_Base PUBLIC ntdll)
endif()
AddDefaultOptionsToTarget(FEXCore_Base)
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
@@ -308,7 +300,6 @@ add_library(JemallocLibs STATIC Utils/AllocatorHooks.cpp)
if (ENABLE_FEX_ALLOCATOR)
target_compile_definitions(JemallocLibs PRIVATE ENABLE_FEX_ALLOCATOR=1)
target_link_libraries(JemallocLibs PUBLIC rpmalloc)
target_include_directories(JemallocLibs PRIVATE "${PROJECT_SOURCE_DIR}/include/")
endif()
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
set_source_files_properties(Interface/HLE/Thunks/Thunks.cpp PROPERTIES COMPILE_DEFINITIONS ENABLE_JEMALLOC_GLIBC=1)
+14 -19
View File
@@ -18,7 +18,7 @@ struct BitSet final {
constexpr static size_t MinimumSize = sizeof(ElementType);
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
ElementType* Memory {};
ElementType* Memory;
void Allocate(size_t Elements) {
size_t AllocateSize = ToBytes(Elements);
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
@@ -33,15 +33,14 @@ struct BitSet final {
FEXCore::Allocator::free(Memory);
Memory = nullptr;
}
[[nodiscard]]
bool Get(T Element) const {
bool Get(T Element) {
return (Memory[Element / MinimumSizeBits] & (1ULL << (Element % MinimumSizeBits))) != 0;
}
void Set(T Element) {
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
}
void Clear(T Element) {
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
}
void MemClear(size_t Elements) {
memset(Memory, 0, ToBytes(Elements));
@@ -49,15 +48,13 @@ struct BitSet final {
void MemSet(size_t Elements) {
memset(Memory, 0xFF, ToBytes(Elements));
}
[[nodiscard]]
static size_t ToBytes(size_t Elements) {
return AlignUp(Elements, MinimumSizeBits) / 8;
uint32_t ToBytes(size_t Elements) {
return AlignUp(Elements, MinimumSizeBits) / MinimumSize;
}
// This very explicitly doesn't let you take an address
// Is only a getter
[[nodiscard]]
bool operator[](T Element) const {
bool operator[](T Element) {
return Get(Element);
}
};
@@ -65,37 +62,35 @@ struct BitSet final {
template<typename T>
struct BitSetView final {
using ElementType = T;
constexpr static size_t MinimumSize = BitSet<T>::MinimumSize;
constexpr static size_t MinimumSizeBits = BitSet<T>::MinimumSizeBits;
constexpr static size_t MinimumSize = sizeof(ElementType);
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
ElementType* Memory {};
ElementType* Memory;
void GetView(BitSet<T>& Set, uint64_t ElementOffset) {
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
}
[[nodiscard]]
bool Get(T Element) const {
bool Get(T Element) {
return (Memory[Element / MinimumSizeBits] & (1ULL << (Element % MinimumSizeBits))) != 0;
}
void Set(T Element) {
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
}
void Clear(T Element) {
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
}
void MemClear(size_t Elements) {
memset(Memory, 0, BitSet<T>::ToBytes(Elements));
memset(Memory, 0, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
}
void MemSet(size_t Elements) {
memset(Memory, 0xFF, BitSet<T>::ToBytes(Elements));
memset(Memory, 0xFF, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
}
// This very explicitly doesn't let you take an address
// Is only a getter
[[nodiscard]]
bool operator[](T Element) const {
bool operator[](T Element) {
return Get(Element);
}
};
+7 -2
View File
@@ -15,9 +15,14 @@ JITSymbols::~JITSymbols() {
}
}
void JITSymbols::InitFile(uint32_t ProcessPID) {
void JITSymbols::InitFile() {
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", ProcessPID);
#ifdef __ANDROID__
// Android simpleperf looks in /data/local/tmp instead of /tmp
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
#else
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
#endif
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
}
+1 -1
View File
@@ -35,7 +35,7 @@ public:
JITSymbols();
~JITSymbols();
void InitFile(uint32_t ProcessPID);
void InitFile();
void RegisterNamedRegion(const void* HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void* HostAddr, uint32_t CodeSize);
-1
View File
@@ -4,7 +4,6 @@
#include <concepts>
#include <string_view>
#include <cstdlib>
namespace FEXCore::StrConv {
template<std::integral T>
+2 -59
View File
@@ -1,10 +1,9 @@
// SPDX-License-Identifier: MIT
#include "Common/StringConv.h"
#include "Utils/Config.h"
#include "FEXCore/Utils/EnumUtils.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/FileLoading.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/StringUtils.h>
@@ -38,16 +37,8 @@ namespace detail {
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
#include <FEXCore/Config/ConfigValues.inl>
constexpr static std::array<std::string_view, FEXCore::Config::ConfigOption::CONFIG_MAX> option_names = {
#define OPT_BASE(type, group, enum, json, default) #json,
#include <FEXCore/Config/ConfigValues.inl>
};
} // namespace detail
std::string_view GetConfigJSONName(FEXCore::Config::ConfigOption option) {
return FEXCore::Config::detail::option_names[option];
}
enum Paths {
PATH_DATA_DIR_LOCAL = 0,
PATH_DATA_DIR_GLOBAL,
@@ -56,7 +47,6 @@ enum Paths {
PATH_CONFIG_FILE_LOCAL,
PATH_CONFIG_FILE_GLOBAL,
PATH_CONFIG_TELEMETRY_FOLDER,
PATH_CACHE_DIR,
PATH_LAST,
};
static std::array<fextl::string, Paths::PATH_LAST> Paths;
@@ -73,10 +63,6 @@ void SetConfigFileLocation(const std::string_view Path, bool Global) {
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
}
void SetCacheDirectory(const std::string_view Path) {
Paths[PATH_CACHE_DIR] = Path;
}
const fextl::string& GetTelemetryDirectory() {
auto& Path = Paths[PATH_CONFIG_TELEMETRY_FOLDER];
if (Path.empty()) {
@@ -104,10 +90,6 @@ const fextl::string& GetConfigFileLocation(bool Global) {
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
}
const fextl::string& GetCacheDirectory() {
return Paths[PATH_CACHE_DIR];
}
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
fextl::string ConfigFile = GetConfigDirectory(Global);
@@ -270,7 +252,7 @@ void Load() {
}
}
static fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
if (PathName.empty()) {
return {};
}
@@ -519,43 +501,4 @@ void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArray
}
}
template void Value<StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
#define CONFIG_AFFECTSCODEGEN
#include <FEXCore/Config/ConfigOptions.inl>
fextl::string SerializeForCache() {
fextl::string Config {};
auto append_string_triple = [](fextl::string& Config, std::string_view Key, ConfigOption Option, auto Value) {
Config.append(Key);
Config.append(1, '\0');
Config.append(fextl::fmt::format("{}", FEXCore::ToUnderlying(Option)));
Config.append(1, '\0');
Config.append(fextl::fmt::format("{}", Value));
Config.append(1, '\0');
};
const auto SerializeValue = [&Config, append_string_triple]<typename T, ConfigOption Option>(auto ConfigVal, const auto Default) {
if (!Config_AffectsCodeGen[FEXCore::ToUnderlying(Option)]) {
// Skip everything that the config says doesn't affect codegen.
return;
}
append_string_triple(Config, FEXCore::Config::GetConfigJSONName(Option), Option, ConfigVal());
};
#define OPT_BASE(type, group, enum, json, default) \
SerializeValue.template operator()<type, CONFIG_##enum>(FEXCore::Config::Get_##enum(), default);
#define OPT_STR(group, enum, json, default) \
SerializeValue.template operator()<fextl::string, CONFIG_##enum>(FEXCore::Config::Get_##enum(), default);
#define OPT_STRARRAY(group, enum, json, default) // Unsupported.
#define OPT_STRENUM(group, enum, json, default) // Unsupported.
#include <FEXCore/Config/ConfigValues.inl>
return Config;
}
FEX_DEFAULT_VISIBILITY bool CheckConfigMatches(std::string_view Config) {
// Serialize current config and just check if it matches.
return SerializeForCache() == Config;
}
} // namespace FEXCore::Config
+5 -182
View File
@@ -4,7 +4,6 @@
"Multiblock": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "true",
"Desc": [
"Controls multiblock code compilation",
"Can cause long JIT compilation times and stutter"
@@ -13,7 +12,6 @@
"MaxInst": {
"Type": "int32",
"Default": "5000",
"AffectsCodeGen": "true",
"Desc": [
"Maximum number of instruction to store in a block"
]
@@ -21,23 +19,13 @@
"EnableCodeCachingWIP": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "true",
"Desc": [
"Enable the code caching subsystem"
]
},
"EnableLazyCodeCachingWIP": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Enable lazy loading of chunks in code caches"
]
},
"EnableCodeCacheValidation": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Enable expensive validation when loading code caches"
]
@@ -45,8 +33,6 @@
"HostFeatures": {
"Type": "strenum",
"Default": "FEXCore::Config::HostFeatures::OFF",
"AffectsCodeGen": "true",
"Comment": "Technically affects codegen, but this is serialized elsewhere.",
"Enums": {
"ENABLESVE": "enablesve",
"DISABLESVE": "disablesve",
@@ -91,11 +77,7 @@
"ENABLESSE4A": "enablesse4a",
"DISABLESSE4A": "disablesse4a",
"ENABLEMOPS": "enablemops",
"DISABLEMOPS": "disablemops",
"ENABLEI8MM": "enablei8mm",
"DISABLEI8MM": "disablei8mm",
"ENABLEDOTPROD": "enabledotprod",
"DISABLEDOTPROD": "disabledotprod"
"DISABLEMOPS": "disablemops"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
@@ -120,15 +102,12 @@
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it",
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it",
"\t{enable,disable}i8mm: Will force enable or disable i8mm even if the host doesn't support it",
"\t{enable,disable}dotprod: Will force enable or disable dotprod even if the host doesn't support it"
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it"
]
},
"SmallTSCScale": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "true",
"Desc": [
"Scales the cycle counter on systems that have low frequencies."
]
@@ -136,115 +115,22 @@
"HideHybrid": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Desc": [
"Hides hybrid CPU core arrangement."
]
},
"SoftwareRNG": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "true",
"Desc": [
"Emulates RDRAND and RDSEED in software when the host does not implement FEAT_RNG."
]
},
"CPUFeatureRegisters": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false",
"Comment": "Technically affects codegen, but this is serialized in to HostFeatures.",
"Desc": [
"Allows overriding cpu feature flags for manual testing"
]
},
"DiskCache": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Enables disk caching for code blocks"
]
},
"DiskCacheFileMapping": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Desc": [
"Maps cache files for faster reading"
]
},
"DiskCacheValidation": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Debug mode that does nothing but validate code hits"
]
},
"DiskCacheRelocationFilter": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Desc": [
"Don't cache blocks with relocations pointing outside of any known region"
]
},
"DiskCacheAnonCaching": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Desc": [
"Attempt to cache anonymous code"
]
},
"DiskCacheMemorySize": {
"Type": "uint32",
"Default": "0",
"AffectsCodeGen": "false",
"Desc": [
"How much memory, if any, to use as a cache for the cache"
]
},
"DiskCachePath": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"Optional base directory override for disk cache"
]
},
"DiskCacheRODBNames": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"Optional list of extra read-only disk cache DBs to consider"
]
},
"DiskCacheMaxFileSize": {
"Type": "uint64",
"Default": "1073741824",
"AffectsCodeGen": "false",
"Desc": [
"Size limit on the main cache file - index not included. Default 1G"
]
},
"DiskCachePruneStaleEntries": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Desc": [
"Allows FEX to prune stale disk cache entries on startup.",
"Frees space by removing cache entries associated with old FEX versions."
]
}
},
"Emulation": {
"RootFS": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
@@ -259,7 +145,6 @@
"ThunkHostLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
"AffectsCodeGen": "false",
"Desc": [
"Folder to find the host-side thunking libraries."
]
@@ -267,7 +152,6 @@
"ThunkGuestLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
"AffectsCodeGen": "false",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
@@ -275,7 +159,6 @@
"ThunkConfig": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"A json file specifying where to overlay the thunks.",
"This can be a filesystem path",
@@ -290,7 +173,6 @@
"Env": {
"Type": "strarray",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"Adds an environment variable to the emulated environment."
]
@@ -298,7 +180,6 @@
"HostEnv": {
"Type": "strarray",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"Adds an environment variable to the host environment.",
"This can be useful for setting environment variables that thunks can pick up.",
@@ -308,7 +189,6 @@
"AdditionalArguments": {
"Type": "strarray",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"Allows the user to pass additional arguments to the application"
]
@@ -316,7 +196,6 @@
"DisableL2Cache": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Desc": [
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
"Can potentially introduce more stutters."
@@ -325,7 +204,6 @@
"DynamicL1Cache": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Desc": [
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
"Can potentially introduce more stutters."
@@ -334,7 +212,6 @@
"DynamicL1CacheIncreaseCountHeuristic": {
"Type": "uint64",
"Default": "250",
"AffectsCodeGen": "false",
"Desc": [
"Threshold of lookups per second that the L1 dynamic cache should increase its size.",
"Lower numbers means more aggressive scaling upward to the maximum size.",
@@ -346,7 +223,6 @@
"DynamicL1CacheDecreaseCountHeuristic": {
"Type": "uint64",
"Default": "50",
"AffectsCodeGen": "false",
"Desc": [
"Threshold of lookups per second that the L1 dynamic cache should decrease its size.",
"The higher the number, the more aggressively it reduces the L1 cache size.",
@@ -360,7 +236,6 @@
"SingleStep": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "true",
"Desc": [
"Single stepping configuration."
]
@@ -368,7 +243,6 @@
"GdbServer": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "true",
"Desc": [
"Enables the GDB server."
]
@@ -376,7 +250,6 @@
"DumpIR": {
"Type": "str",
"Default": "no",
"AffectsCodeGen": "false",
"Desc": [
"Folder to dump the IR in to.",
"[no, stdout, stderr, server, <Folder>]"
@@ -385,7 +258,6 @@
"PassManagerDumpIR": {
"Type": "strenum",
"Default": "FEXCore::Config::PassManagerDumpIR::OFF",
"AffectsCodeGen": "false",
"Enums": {
"BEFOREOPT": "beforeopt",
"AFTEROPT": "afteropt",
@@ -404,7 +276,6 @@
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"When the test harness ends, print the GPR state."
]
@@ -412,7 +283,6 @@
"O0": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "true",
"Desc": [
"Disables optimizations passes for debugging."
]
@@ -420,7 +290,6 @@
"GlobalJITNaming": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Uses JITSymbols to name all JIT state as one symbol",
"Useful for querying how much time is spent inside of the JIT",
@@ -430,7 +299,6 @@
"LibraryJITNaming": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Uses JITSymbols to name JIT symbols grouped by library",
"Useful for querying how much time is spent in each guest library",
@@ -440,7 +308,6 @@
"BlockJITNaming": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Uses JITSymbols to name JIT symbols",
"Useful for determining hot blocks of code",
@@ -450,7 +317,6 @@
"GDBSymbols": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Integrates with GDB using the JIT interface.",
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
@@ -461,7 +327,6 @@
"InjectLibSegFault": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Sets the environment variable LD_PRELOAD=libSegFault.so",
"This allows the user to very easily enable libSegFault without dealing with environment variables",
@@ -473,7 +338,6 @@
"Disassemble": {
"Type": "strenum",
"Default": "FEXCore::Config::Disassemble::OFF",
"AffectsCodeGen": "false",
"Enums": {
"DISPATCHER": "dispatcher",
"BLOCKS": "blocks",
@@ -490,7 +354,6 @@
"X86Disassemble": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Enables x86/x86-64 guest disassembly output for compiled blocks.",
"Requires FEX to be built with -DENABLE_ZYDIS=TRUE"
@@ -499,7 +362,6 @@
"ForceSVEWidth": {
"Type": "uint32",
"Default": "0",
"AffectsCodeGen": "true",
"Desc": [
"Allows overriding the SVE width in the vixl simulator.",
"Useful as a debugging feature."
@@ -508,7 +370,6 @@
"DisableTelemetry": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "true",
"Desc": [
"Disables telemetry at runtime.",
"Useful for CI instcountCI mostly"
@@ -519,7 +380,6 @@
"SilentLog": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Desc": [
"Disables logging"
]
@@ -527,7 +387,6 @@
"OutputLog": {
"Type": "str",
"Default": "server",
"AffectsCodeGen": "false",
"Desc": [
"File to write FEX output to.",
"[stderr, server, <Filename>]"
@@ -536,7 +395,6 @@
"TelemetryDirectory": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"Redirects the telemetry folder that FEX usually writes to.",
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/fex-emu/Telemetry/}"
@@ -545,7 +403,6 @@
"ProfileStats": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Enables FEX's low-overhead sampling profile statistics.",
"Requires a supported version of Mangohud to see the results"
@@ -554,7 +411,6 @@
"EnableGpuvisProfiling": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Enables profiling when FEX was built with the gpuvis profiler backend."
]
@@ -564,7 +420,6 @@
"SMCChecks": {
"Type": "uint8",
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
"AffectsCodeGen": "true",
"TextDefault": "mtrack",
"ArgumentHandler": "SMCCheckHandler",
"Desc": [
@@ -577,7 +432,6 @@
"TSOEnabled": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "true",
"Desc": [
"Controls TSO IR ops.",
"Highly likely to break any multithreaded application if disabled."
@@ -586,7 +440,6 @@
"VectorTSOEnabled": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "true",
"Desc": [
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
]
@@ -594,7 +447,6 @@
"MemcpySetTSOEnabled": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "true",
"Desc": [
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
"Only affects REP MOVS and REP STOS instructions"
@@ -603,7 +455,6 @@
"HalfBarrierTSOEnabled": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "true",
"Desc": [
"When TSO emulation is enabled, controls if unaligned loads and stores should be backpatched to half-barrier atomics.",
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
@@ -612,7 +463,6 @@
"StrictInProcessSplitLocks": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Strict global lock when handling an unaligned atomic that crosses a 16-byte or cacheline granularity",
"This is required to ensure a split-lock doesn't tear inside the process"
@@ -621,7 +471,6 @@
"KernelUnalignedAtomicBackpatching": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Desc": [
"When the kernel unaligned atomic handler is enabled, use backpatching to reduce kernel context switches."
]
@@ -629,7 +478,6 @@
"VolatileMetadata": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "true",
"Desc": [
"Use volatile metadata in PE files to inform TSO instructions when available.",
"When metadata is unavailable falls back to the currently enabled TSO options."
@@ -638,7 +486,6 @@
"X87ReducedPrecision": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "true",
"Desc": [
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
]
@@ -646,7 +493,6 @@
"StallProcess": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Forces a process to stall out on initialization",
"Useful for a process that keeps restarting and doesn't work"
@@ -655,7 +501,6 @@
"HideHypervisorBit": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Hides the hypervisor CPUID bit when set.",
"Should only be used for applications that have issues with this set."
@@ -664,7 +509,6 @@
"StartupSleep": {
"Type": "uint32",
"Default": "0",
"AffectsCodeGen": "false",
"Desc": [
"Sleeps the process at startup for a duration of seconds.",
"Useful if an application crashes too quickly to attach a debugger."
@@ -673,7 +517,6 @@
"StartupSleepProcName": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"Contrains the startup sleep to only apply to processes that match this name."
]
@@ -681,7 +524,6 @@
"MonoHacks": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "true",
"Desc": [
"Permits a hook-based SMC approach and smaller JIT blocks when mono is detected."
]
@@ -691,7 +533,6 @@
"ServerSocketPath": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"Override for a FEXServer socket path. Only useful for chroots."
]
@@ -699,7 +540,6 @@
"NeedsSeccomp": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Desc": [
"Disables inline syscalls in order to support seccomp handling"
]
@@ -707,7 +547,6 @@
"ExtendedVolatileMetadata": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "true",
"Desc": [
"Configuration provided volatile metadata. Only implemented for WoW64/arm64ec.",
"Limited in its use but can be handy.",
@@ -732,18 +571,15 @@
"Misc": {
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false"
"Default": "false"
},
"APP_FILENAME": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false"
"Default": ""
},
"APP_CONFIG_NAME": {
"Type": "str",
"Default": "",
"AffectsCodeGen": "false",
"Desc": [
"This is the application config name that has been loaded.",
"This differs from APP_FILENAME in two ways",
@@ -754,29 +590,16 @@
},
"IS64BIT_MODE": {
"Type": "bool",
"Default": "false",
"AffectsCodeGen": "false",
"Comment": "Technically affects codegen, but this is serialized elsewhere."
"Default": "false"
},
"DISABLE_VIXL_INDIRECT_RUNTIME_CALLS": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "false",
"Comment": "Technically affects codegen, but only shows up in the test harness.",
"Desc": [
"This option is used for the InstructionCountCI so it can generate the same codegen between Arm64 hosts and vixl simulator hosts.",
"Vixl simulator indirect runtime calls are a special hlt instruction with metadata after it. Effectively making a custom call instruction.",
"With visual simulator calls disabled, the code generation would be the same as on a native Arm64 host, but running the code is broken."
]
},
"CONFIG_VERSION": {
"Type": "uint32",
"Default": "0",
"AffectsCodeGen": "true",
"Comment": [
"Meta option that if config has ever changed definitions dramatically enough that we can rev the version.",
"Be mindful that this will invalidate all caches!"
]
}
}
}
+1 -7
View File
@@ -53,12 +53,6 @@ FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionN
}
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const {
return Thread->CPUBackend->IsAddressInCodeBuffer(Address) || CodeCache.IsAddressInMappedCodeBuffer(Address);
return Thread->CPUBackend->IsAddressInCodeBuffer(Address);
}
bool FEXCore::Context::ContextImpl::RequiresRelocatableConstants() const {
// Support relocation when generating a cache or when generating reference code for validation
return CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHEVALIDATION() || DiskCache.IsWritingDiskCache();
}
} // namespace FEXCore::Context
+11 -129
View File
@@ -4,8 +4,6 @@
#include "Common/JitSymbols.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/DiskCache.h"
#include "Interface/Core/SharedCodeBufferManager.h"
#include <Interface/IR/IntrusiveIRList.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
@@ -66,8 +64,6 @@ struct CustomIRResult {
using BlockDelinkerFunc = void (*)(FEXCore::Context::ExitFunctionLinkData* Record);
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
constexpr static bool BLOCK_DEBUGGING = false;
class CodeCache : public AbstractCodeCache {
public:
CodeCache(ContextImpl&);
@@ -80,16 +76,11 @@ public:
bool IsGeneratingCache = false;
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
FEX_CONFIG_OPT(EnableLazyCodeCaching, ENABLELAZYCODECACHINGWIP);
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
uint64_t ComputeCodeMapId(std::string_view Filename, int FD) override;
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
fextl::unique_ptr<MappedCodeCacheFile> LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo&, uint64_t FileStartVA) override;
bool EnableLoadedSection(Core::InternalThreadState*, MappedCodeCacheFile&, const ExecutableFileSectionInfo&) override;
void FinalizeCodePages(MappedCodeCacheFile&, std::span<std::byte> CodeRange) override;
bool LoadData(Core::InternalThreadState*, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
/**
* Performs expensive extra validation on the loaded code cache data.
@@ -127,14 +118,9 @@ public:
*/
[[nodiscard]]
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations, bool ForStorage);
// Same but on disk cache packed relocations
[[nodiscard]]
bool ApplyPackedCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const DiskCache::BlobSmallRelocation> SmallRelocs,
std::span<const DiskCache::BlobThunkRelocation> ThunkRelocs);
};
class ContextImpl final : public FEXCore::Context::Context, public CPU::SharedCodeBufferManager {
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
public:
// Context base class implementation.
bool InitCore() override;
@@ -159,32 +145,32 @@ public:
void SetXMMRegistersFromState(FEXCore::Core::InternalThreadState* Thread, const __uint128_t* XMM_Low, const __uint128_t* YMM_High) override;
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread.
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param InitialRIP The starting RIP of this thread
* @param StackPointer The starting RSP of this thread
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* Parent thread Creation:
* - Thread = CreateThread();
* - Thread->CurrentFrame->State.rip = InitialRIP;
* - Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP] = InitialStack;
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
* - CTX->ExecuteThread(Thread);
* OS thread Creation:
* - Thread = CreateThread(NewState);
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
* - ThreadHandler calls `CTX->ExecuteThread(Thread)`
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(CopyOfThreadState);
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
* - ExecuteThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(NewState);
* - Thread = CreateThread(0, 0, NewState, PPID);
* - HandleCallback(Thread, RIP);
*/
FEXCore::Core::InternalThreadState* CreateThread(const FEXCore::Core::CPUState* NewThreadState) override;
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState) override;
/**
* @brief Destroys this FEX thread object and stops tracking it internally
@@ -205,8 +191,6 @@ public:
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
virtual void InitDiskCache() override {}
CodeCache& GetCodeCache() override {
return CodeCache;
}
@@ -248,100 +232,7 @@ public:
}
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
std::atomic<uint64_t>& GetMonoBackPatcherBlock() {
return MonoBackpatcherBlock;
}
// Manual debugging tooling which is useful for developers.
struct TrackingEmpty {
// RIP stepping handling
virtual void AddSingleStepTarget(uint64_t GuestRIP) {}
virtual void AddSingleStepTargetRange(uint64_t RIPBegin, uint64_t RipEnd) {}
virtual void AllTargetSingleStep() {}
virtual void RemoveSingleStepTarget(uint64_t GuestRIP) {}
virtual bool IsSingleStepTarget(uint64_t GuestRIP) {
return false;
}
// Watchpoints
virtual void AddWriteWatchPoint(uint64_t Ptr) {}
virtual void AddReadWatchPoint(uint64_t Ptr) {}
virtual bool ContainsWriteWatchPoint(uint64_t Ptr, size_t Size) {
return false;
}
virtual bool ContainsReadWatchPoint(uint64_t Ptr, size_t Size) {
return false;
}
};
struct TrackingPossible final : public TrackingEmpty {
void AddSingleStepTarget(uint64_t GuestRIP) override {
SingleStepTargets.emplace(GuestRIP);
}
virtual void AddSingleStepTargetRange(uint64_t RIPBegin, uint64_t RIPEnd) override {
SingleStepRanges.emplace_back(Range {RIPBegin, RIPEnd});
}
void RemoveSingleStepTarget(uint64_t GuestRIP) override {
SingleStepTargets.erase(GuestRIP);
}
void AllTargetSingleStep() override {
SingleStepEverything = true;
}
bool IsSingleStepTarget(uint64_t GuestRIP) override {
return SingleStepEverything || SingleStepTargets.contains(GuestRIP) || IsInRange(GuestRIP);
}
void AddWriteWatchPoint(uint64_t Ptr) override {
WatchWriteTargets.emplace(Ptr);
}
void AddReadWatchPoint(uint64_t Ptr) override {
WatchReadTargets.emplace(Ptr);
}
bool ContainsWriteWatchPoint(uint64_t Ptr, size_t Size) override {
return ContainsRange(WatchWriteTargets, Ptr, Size);
}
bool ContainsReadWatchPoint(uint64_t Ptr, size_t Size) override {
return ContainsRange(WatchReadTargets, Ptr, Size);
}
private:
bool SingleStepEverything {};
fextl::set<uint64_t> SingleStepTargets {};
fextl::set<uint64_t> WatchWriteTargets {};
fextl::set<uint64_t> WatchReadTargets {};
struct Range {
uint64_t Begin, End;
};
fextl::vector<Range> SingleStepRanges {};
bool IsInRange(uint64_t RIP) const {
return std::ranges::any_of(SingleStepRanges, [RIP](const auto& range) { return RIP >= range.Begin && RIP <= range.End; });
}
static bool ContainsRange(const fextl::set<uint64_t>& Set, uint64_t Ptr, size_t Size) {
for (auto it = Set.lower_bound(Ptr); it != Set.end(); --it) {
auto Watch = *it;
if (Watch < Ptr) {
break;
}
if (Watch >= Ptr && Watch < (Ptr + Size)) {
return true;
}
}
return false;
}
};
using TrackingStructure = std::conditional<BLOCK_DEBUGGING, TrackingPossible, TrackingEmpty>::type;
TrackingStructure BlockDebuggerTracker {};
public:
struct {
uint64_t VirtualMemSize {1ULL << 36};
@@ -368,16 +259,10 @@ public:
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
FEX_CONFIG_OPT(SoftwareRNG, SOFTWARERNG);
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
} Config;
bool SoftwareRNGEnabled() const {
return Config.SoftwareRNG() && HostRNGAvailable;
}
bool HostRNGAvailable {};
FEXCore::Utils::WritePriorityMutex::Mutex CodeInvalidationMutex {};
uint32_t StrictSplitLockMutex {};
@@ -389,7 +274,6 @@ public:
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
FEXCore::ThunkHandler* ThunkHandler {};
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
DiskCache::DiskCache DiskCache;
CodeCache CodeCache;
fextl::unique_ptr<CodeMapWriter> CodeMapWriter;
@@ -467,8 +351,6 @@ public:
return Config.MonoHacks && MonoDetected;
}
bool RequiresRelocatableConstants() const;
protected:
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
@@ -360,7 +360,6 @@ namespace x32 {
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr, size_t size)
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
, EmitterCTX {ctx}
, SupportCodeRelocations {ctx->RequiresRelocatableConstants()}
#ifdef VIXL_SIMULATOR
, Simulator {&SimDecoder, stdout, vixl::aarch64::SimStack(SimulatorStackSize).Allocate()}
#endif
@@ -426,7 +425,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
NOPPad = false;
} else if (Pad == PadType::AUTOPAD) {
// Force NOP padding to ensure relocated constants always have enough encoding space available
NOPPad = SupportCodeRelocations;
NOPPad = EnableCodeCaching;
}
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
@@ -587,8 +586,8 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
}};
for (const auto& [rt, rt2] : CalleeSaved) {
stp<ARMEmitter::IndexType::PRE>(rt, rt2, ARMEmitter::Reg::rsp, -16);
for (auto& RegPair : CalleeSaved) {
stp<ARMEmitter::IndexType::PRE>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, -16);
}
// Additionally we need to store the lower 64bits of v8-v15
@@ -605,8 +604,9 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
// We just saved x19 so it is safe
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r19, ARMEmitter::Reg::rsp, 0);
for (const auto& [rt, rt2, rt3, rt4] : FPRs) {
st4(ARMEmitter::SubRegSize::i64Bit, rt, rt2, rt3, rt4, 0, ARMEmitter::Reg::r19, 32);
for (auto& RegQuad : FPRs) {
st4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
ARMEmitter::Reg::r19, 32);
}
}
@@ -616,8 +616,9 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
}};
for (const auto& [rt, rt2, rt3, rt4] : FPRs) {
ld4(ARMEmitter::SubRegSize::i64Bit, rt, rt2, rt3, rt4, 0, ARMEmitter::Reg::rsp, 32);
for (auto& RegQuad : FPRs) {
ld4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
ARMEmitter::Reg::rsp, 32);
}
constexpr static std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
@@ -629,12 +630,12 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
}};
for (const auto& [rt, rt2] : CalleeSaved) {
ldp<ARMEmitter::IndexType::POST>(rt, rt2, ARMEmitter::Reg::rsp, 16);
for (auto& RegPair : CalleeSaved) {
ldp<ARMEmitter::IndexType::POST>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, 16);
}
}
void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, const FillSpecialRegsOptions& Options) {
void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs) {
#ifndef VIXL_SIMULATOR
if (EmitterCTX->HostFeatures.SupportsAFP) {
// Enable AFP features when filling JIT state.
@@ -650,7 +651,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
(1U << 2) | // NEP
(1U << 1)); // AH
if (Options.SetFIZ) {
if (SetFIZ) {
// Insert MXCSR.DAZ in to FIZ
ldr(TmpReg2.W(), STATE.R(), offsetof(FEXCore::Core::CPUState, mxcsr));
bfxil(ARMEmitter::Size::i64Bit, TmpReg, TmpReg2, 6, 1);
@@ -660,7 +661,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
}
#endif
if (Options.SetPredRegs && EmitterCTX->HostFeatures.SupportsSVE()) {
if (SetPredRegs && (EmitterCTX->HostFeatures.SupportsSVE256 || EmitterCTX->HostFeatures.SupportsSVE128)) {
// Set up predicate registers.
// We don't bother spilling these in SpillStaticRegs,
// since all that matters is we restore them on a fill.
@@ -823,7 +824,7 @@ void Arm64Emitter::FillStaticRegs(FillStaticRegOptions Options) {
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
}
FillSpecialRegs(TmpReg, TmpReg2, {.SetFIZ = true, .SetPredRegs = Options.FPRs});
FillSpecialRegs(TmpReg, TmpReg2, true, Options.FPRs);
if (Options.FPRs) {
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
@@ -1060,7 +1061,6 @@ size_t Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, boo
SpillStaticRegs(TmpReg, {
.GPRSpillMask = PreserveSRAMask,
.FPRSpillMask = PreserveSRAFPRMask,
.FPRs = FPRs,
});
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
@@ -1123,7 +1123,6 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
void Arm64Emitter::Align16B() {
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
LOGMAN_THROW_A_FMT((CurrentOffset & 3) == 0, "Can't Align16B code that isn't 4-byte aligned!");
for (uint64_t i = (-CurrentOffset & 0xF); i != 0; i -= 4) {
nop();
}
@@ -129,21 +129,7 @@ protected:
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
uint32_t PairRegisters = 0;
bool SupportCodeRelocations;
struct FillSpecialRegsOptions {
// Whether or not to set the FPCR.FIZ (flush inputs to zero) bit in the FPCR to
// the current value of the emulated MXCSR.DAZ bit.
// Will only attempt to do so, even when set to true, if and only if the host system
// supports FEAT_AFP.
bool SetFIZ {};
// Whether or not FillSpecialRegs should load our SVE predicate temporaries
// with certain canned values that accelerate some operations. Will (obviously)
// not load predicates, even if set to true, on host systems that do not support SVE.
bool SetPredRegs {};
};
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, const FillSpecialRegsOptions& Options);
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
// Correlate an ARM register back to an x86 register index.
// Returning REG_INVALID if there was no mapping.
@@ -322,6 +308,8 @@ protected:
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
#endif
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
};
} // namespace FEXCore::CPU
+109 -13
View File
@@ -11,12 +11,20 @@
#include <cstdint>
#ifndef _WIN32
#include <sys/prctl.h>
#endif
namespace FEXCore {
namespace CPU {
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_INCREMENTAL_U8_INDEX
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
@@ -35,8 +43,6 @@ namespace CPU {
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB_UPPER
{0x0706'0504'0302'0100ULL, 0x1716'1514'1312'1110ULL}, // NAMED_VECTOR_256_MID_ELEMENT_SWAP
{0x0F0E'0D0C'0B0A'0908ULL, 0x1F1E'1D1C'1B1A'1918ULL}, // NAMED_VECTOR_256_MID_ELEMENT_SWAP_UPPER
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_ONE
{0xD49A'784B'CD1B'8AFEULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_LOG2_10
{0xB8AA'3B29'5C17'F0BCULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_LOG2_E
@@ -267,9 +273,9 @@ namespace CPU {
return TotalLUT;
}()};
CPUBackend::CPUBackend(SharedCodeBufferManager& SharedCodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
CPUBackend::CPUBackend(CodeBufferManager& CodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
: ThreadState(ThreadState)
, SharedCodeBuffers(SharedCodeBuffers) {
, CodeBuffers(CodeBuffers) {
auto& Ptrs = ThreadState->CurrentFrame->Pointers;
@@ -308,11 +314,11 @@ namespace CPU {
CPUBackend::~CPUBackend() = default;
auto CPUBackend::AcquireNewSharedCodeBuffer() -> CodeBuffer* {
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
auto PrevCodeBuffer = CurrentCodeBuffer;
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer = SharedCodeBuffers.StartLargerCodeBuffer();
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
RegisterForSignalHandler(std::move(PrevCodeBuffer));
return CurrentCodeBuffer.get();
@@ -330,7 +336,7 @@ namespace CPU {
}
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
auto NewCodeBuffer = SharedCodeBuffers.GetLatest();
auto NewCodeBuffer = CodeBuffers.GetLatest();
if (CurrentCodeBuffer != NewCodeBuffer) {
RegisterForSignalHandler(CurrentCodeBuffer);
return std::exchange(CurrentCodeBuffer, NewCodeBuffer);
@@ -338,17 +344,107 @@ namespace CPU {
return nullptr;
}
GuestToHostMap& GetLookupCache(const CodeBuffer& Buffer) {
return *Buffer.LookupCache;
}
CodeBuffer::CodeBuffer(size_t Size)
: AllocatedSize(Size) {
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
FEXCore::Allocator::ProtectOptions::None)) {
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
}
FEXCore::Allocator::VirtualName("FEXMemJIT", reinterpret_cast<void*>(Ptr), Size);
// Huge-pages reduce the amount of iTLB misses dramatically when it works.
FEXCore::Allocator::VirtualTHPControl(reinterpret_cast<void*>(Ptr), Size, FEXCore::Allocator::THPControl::Enable);
LookupCache = fextl::make_unique<GuestToHostMap>();
}
CodeBuffer::~CodeBuffer() {
FEXCore::Allocator::VirtualFree(Ptr, AllocatedSize);
}
auto CodeBufferManager::AllocateNew(size_t Size) -> fextl::shared_ptr<CodeBuffer> {
#ifndef _WIN32
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
//
// MDWE prevents applications from creating RWX memory mappings.
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
//
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
//
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
//
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
// -1: The kernel doesn't support MDWE
// 0: MDWE is supported but disabled
// >0: MDWE is enabled, hence prohibiting RWX mappings
#ifndef PR_GET_MDWE
#define PR_GET_MDWE 66
#endif
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
if (MDWE != -1 && MDWE != 0) {
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
}
#endif
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
Latest = Buffer;
LatestOffset = 0;
OnCodeBufferAllocated(Buffer);
return Buffer;
}
fextl::shared_ptr<CodeBuffer> CodeBufferManager::GetLatest() {
if (!Latest) {
if (FEXCore::Config::Get_ENABLECODECACHINGWIP()) {
// Start with a larger code buffer to avoid resizes that would discard
// code loaded from caches
AllocateNew(MAX_CODE_SIZE);
} else {
AllocateNew(INITIAL_CODE_SIZE);
}
}
return Latest;
}
fextl::shared_ptr<CodeBuffer> CodeBufferManager::StartLargerCodeBuffer() {
if (!Latest) {
// Allocate initial CodeBuffer and return it
return GetLatest();
}
auto NewCodeBufferSize = GetLatest()->AllocatedSize;
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
return AllocateNew(NewCodeBufferSize);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
const auto CheckCodeBuffer = [](const CodeBuffer& Buffer, uintptr_t Address) {
const auto BufferPtr = reinterpret_cast<uintptr_t>(Buffer.GetBufferBase());
const uintptr_t LastPageAddr = BufferPtr + Buffer.UsableSize();
return (Address >= BufferPtr && Address < LastPageAddr);
auto CheckCodeBuffer = [](CodeBuffer& Buffer, uintptr_t Address) {
// The last page of the code buffer is protected, so we need to exclude it from the valid range
// when checking if the address is in the code buffer.
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.AllocatedSize - 1, FEXCore::Utils::FEX_PAGE_SIZE);
return (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr);
};
if (CheckCodeBuffer(*CurrentCodeBuffer, Address)) {
return true;
}
for (const auto& Buffer : SignalHandlerCodeBuffers) {
for (auto& Buffer : SignalHandlerCodeBuffers) {
if (CheckCodeBuffer(*Buffer, Address)) {
return true;
}
+56 -13
View File
@@ -8,8 +8,6 @@ $end_info$
#pragma once
#include "Interface/Core/SharedCodeBufferManager.h"
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
@@ -18,7 +16,6 @@ $end_info$
#include <FEXCore/fextl/map.h>
#include <cstdint>
#include <span>
namespace FEXCore::CPU {
union Relocation;
@@ -44,10 +41,63 @@ namespace CodeSerialize {
struct GuestToHostMap;
namespace CPU {
struct CodeBuffer {
uint8_t* Ptr;
size_t AllocatedSize; // including guard page; see UsableSize()
fextl::unique_ptr<GuestToHostMap> LookupCache;
CodeBuffer(size_t Size);
CodeBuffer(const CodeBuffer&) = delete;
CodeBuffer& operator=(const CodeBuffer&) = delete;
CodeBuffer(CodeBuffer&& oth) = delete;
CodeBuffer& operator=(CodeBuffer&&) = delete;
~CodeBuffer();
/// Returns the number of bytes available for storing code
size_t UsableSize() const {
return AllocatedSize - FEXCore::Utils::FEX_PAGE_SIZE;
}
};
/**
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
*
* The CodeBuffer is managed as a partially persistent data structure:
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
*/
class CodeBufferManager {
public:
// Get the CodeBuffer that was most recently allocated.
// This is the only CodeBuffer that data may be written to.
fextl::shared_ptr<CodeBuffer> GetLatest();
// Allocate a new CodeBuffer with geometric growth up to an internal maximum.
// Subsequent calls to GetLatest will point to the returned buffer.
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer();
// Write offset into the latest CodeBuffer
std::size_t LatestOffset {};
// Protects writes to the latest CodeBuffer and changes to LatestOffset
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
private:
fextl::shared_ptr<CodeBuffer> Latest;
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
};
class CPUBackend {
public:
CPUBackend(SharedCodeBufferManager&, FEXCore::Core::InternalThreadState*);
CPUBackend(CodeBufferManager&, FEXCore::Core::InternalThreadState*);
virtual ~CPUBackend();
@@ -57,8 +107,6 @@ namespace CPU {
fextl::map<uint64_t, uint8_t*> EntryPoints;
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
size_t Size;
// Offset of BlockBegin from the start of the CodeBuffer it lives in
uint64_t HostCodeOffset;
};
// Header that can live at the start of a JIT block.
@@ -118,10 +166,6 @@ namespace CPU {
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
virtual CompiledCode LoadCachedCode(std::span<const uint8_t> HostBytes) {
return {};
}
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) = 0;
virtual void ClearCache() {}
@@ -145,9 +189,8 @@ namespace CPU {
FEXCore::Core::InternalThreadState* ThreadState;
// Acquires a new shared code buffer, setting `CurrentCodeBuffer` and returning a pointer to it.
[[nodiscard]]
CodeBuffer* AcquireNewSharedCodeBuffer();
CodeBuffer* GetEmptyCodeBuffer();
// This is the code buffer containing the main code under execution by this thread.
// CheckCodeBufferUpdate must be used before compiling new code.
@@ -156,7 +199,7 @@ namespace CPU {
// Old CodeBuffer generations required to be valid until returning from signal handlers
fextl::vector<fextl::shared_ptr<CodeBuffer>> SignalHandlerCodeBuffers;
SharedCodeBufferManager& SharedCodeBuffers;
CodeBufferManager& CodeBuffers;
private:
void RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer>);
+75 -93
View File
@@ -100,7 +100,7 @@ namespace ProductNames {
#endif
} // namespace ProductNames
static uint32_t GetCPUID_Syscall() {
uint32_t GetCPUID_Syscall() {
uint32_t CPU {};
FHU::Syscalls::getcpu(&CPU, nullptr);
return CPU;
@@ -148,7 +148,7 @@ uint64_t GetCycleCounterFrequency() {
return Result;
}
static uint32_t GetCPUID_TPIDRRO() {
uint32_t GetCPUID_TPIDRRO() {
uint64_t Result {};
__asm("mrs %[Res], TPIDRRO_EL0" : [Res] "=r"(Result));
return Result;
@@ -316,8 +316,9 @@ void CPUIDEmu::SetupHostHybridFlag() {
// Walk our list of CPUMIDRs to find the most little core
for (size_t j = LowestMIDRIdx; j < CPUMIDRs.size(); ++j) {
const auto& MIDROption = CPUMIDRs[j];
auto& MIDROption = CPUMIDRs[i];
if ((MIDROption.Implementer == Implementer && MIDROption.Part == Part) || (MIDROption.Implementer == 0 && MIDROption.Part == 0)) {
LowestMIDRIdx = j;
LowestMIDR = MIDR;
break;
@@ -458,8 +459,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(GetCPUID() << 24); // Local APIC ID
const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled();
Res.ecx = (1 << 0) | // SSE3
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
(1 << 2) | // DS area supports 64bit layout
@@ -490,13 +489,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
(SupportsAVX() << 27) | // OSXSAVE
(SupportsAVX() << 28) | // AVX
(SupportsAVX() << 29) | // F16C
(SupportsRAND << 30) | // RDRAND
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
(Hypervisor << 31);
Res.edx = (1 << 0) | // FPU
(1 << 1) | // Virtual 8086 mode enhancements
(1 << 2) | // Debugging extensions
(1 << 3) | // Page size extension
(0 << 2) | // Debugging extensions
(0 << 3) | // Page size extension
(1 << 4) | // RDTSC supported
(1 << 5) | // MSR supported
(1 << 6) | // PAE
@@ -650,20 +649,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) const {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
FEXCore::CPUID::FunctionResults Res {};
// AVX-VNNI is only advertised when the CPU supports I8MM or Dot Product.
// Without these features the implementation is so slow that it is likely
// to harm performance.
const uint32_t SupportsAVXVNNI = SupportsAVX() && (CTX->HostFeatures.SupportsI8MM || CTX->HostFeatures.SupportsDotProd);
if (Leaf == 0) {
#ifndef _WIN32
constexpr uint32_t SUPPORTS_RDPID = 1;
#else
// RDPID under WIN32 is only supported if CPUIndex is available in TPIDRRO.
const uint32_t SUPPORTS_RDPID = SupportsCPUIndexInTPIDRRO;
#endif
// Disable Enhanced REP MOVS when TSO is enabled.
// vcruntime140 memmove will use `rep movsb` in this case which completely destroys perf in Hades(appId 1145360)
// This is due to LRCPC performance on Cortex being abysmal.
@@ -672,44 +658,40 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT;
const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled();
// Number of subfunctions
// TODO: For now, subfunction 1 only exposes AVX-VNNI so we make it conditional
// on AVX-VNNI support. We should revisit this if/when we add more to this leaf.
Res.eax = SupportsAVXVNNI;
Res.ebx = (1 << 0) | // FS/GS support
(0 << 1) | // TSC adjust MSR
(0 << 2) | // SGX
(SupportsAVX() << 3) | // BMI1
(0 << 4) | // Intel Hardware Lock Elison
(SupportsAVX() << 5) | // AVX2 support
(1 << 6) | // FPU data pointer updated only on exception
(1 << 7) | // SMEP support
(SupportsAVX() << 8) | // BMI2
(SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB
(1 << 10) | // INVPCID for system software control of process-context
(0 << 11) | // Restricted transactional memory
(0 << 12) | // Intel resource directory technology Monitoring
(1 << 13) | // Deprecates FPU CS and DS
(0 << 14) | // Intel MPX
(0 << 15) | // Intel Resource Directory Technology Allocation
(0 << 16) | // AVX512-F
(0 << 17) | // AVX512-DQ
(SupportsRAND << 18) | // RDSEED
(1 << 19) | // ADCX and ADOX instructions
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
(0 << 21) | // AVX512-IFMA
(0 << 22) | // PCOMMIT (deprecated?)
(1 << 23) | // CLFLUSHOPT instruction
(1 << 24) | // CLWB instruction
(0 << 25) | // Intel processor trace
(0 << 26) | // AVX512-PF
(0 << 27) | // AVX512-ER
(0 << 28) | // AVX512-CD
(Features.SHA << 29) | // SHA instructions
(0 << 30) | // AVX512-BW
(0 << 31); // AVX512-VL
Res.eax = 0x0;
Res.ebx = (1 << 0) | // FS/GS support
(0 << 1) | // TSC adjust MSR
(0 << 2) | // SGX
(SupportsAVX() << 3) | // BMI1
(0 << 4) | // Intel Hardware Lock Elison
(SupportsAVX() << 5) | // AVX2 support
(1 << 6) | // FPU data pointer updated only on exception
(1 << 7) | // SMEP support
(SupportsAVX() << 8) | // BMI2
(SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB
(1 << 10) | // INVPCID for system software control of process-context
(0 << 11) | // Restricted transactional memory
(0 << 12) | // Intel resource directory technology Monitoring
(1 << 13) | // Deprecates FPU CS and DS
(0 << 14) | // Intel MPX
(0 << 15) | // Intel Resource Directory Technology Allocation
(0 << 16) | // AVX512-F
(0 << 17) | // AVX512-DQ
(CTX->HostFeatures.SupportsRAND << 18) | // RDSEED
(1 << 19) | // ADCX and ADOX instructions
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
(0 << 21) | // AVX512-IFMA
(0 << 22) | // PCOMMIT (deprecated?)
(1 << 23) | // CLFLUSHOPT instruction
(1 << 24) | // CLWB instruction
(0 << 25) | // Intel processor trace
(0 << 26) | // AVX512-PF
(0 << 27) | // AVX512-ER
(0 << 28) | // AVX512-CD
(Features.SHA << 29) | // SHA instructions
(0 << 30) | // AVX512-BW
(0 << 31); // AVX512-VL
Res.ecx = (1 << 0) | // PREFETCHWT1
(0 << 1) | // AVX512VBMI
@@ -733,7 +715,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 19) | // MPX MAWAU
(0 << 20) | // MPX MAWAU
(0 << 21) | // MPX MAWAU
(SUPPORTS_RDPID << 22) | // RDPID Read Processor ID
(1 << 22) | // RDPID Read Processor ID
(0 << 23) | // AES Key Locker
(1 << 24) | // bus-lock-detect
(0 << 25) | // CLDEMOTE
@@ -777,38 +759,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 30) | // Arch capabilities - MSR module specific
(0 << 31); // SSBD - Speculative Store Bypass Disable
} else if (Leaf == 1) {
Res.eax = (0U << 0) | // SHA512
(0U << 1) | // SM3
(0U << 2) | // SM4
(0U << 3) | // RAO_INT
(SupportsAVXVNNI << 4) | // AVX_VNNI
(0U << 5) | // AVX512_BF16
(0U << 6) | // LASS (Linear Address Space Separation)
(0U << 7) | // CMPCCXADD
(0U << 8) | // ARCH_PERFMON_EXT
(0U << 9) | // Reserved
(0U << 10) | // FAST_REP_MOVSB
(0U << 11) | // FAST_REP_STOSB
(0U << 12) | // FAST_REP_CMPSB_SCASB
(0U << 13) | // Reserved
(0U << 14) | // Reserved
(0U << 15) | // Reserved
(0U << 16) | // Reserved
(0U << 17) | // FRED (Flexible Return and Event Delivery)
(0U << 18) | // LKGS (Load into Kernel GS Base)
(0U << 19) | // WRMSRNS
(0U << 20) | // NMI_SRC
(0U << 21) | // AMX_FP16
(0U << 22) | // HRESET
(0U << 23) | // AVX_IFMA
(0U << 24) | // Reserved
(0U << 25) | // Reserved
(0U << 26) | // LAM (Linear Address Masking)
(0U << 27) | // MSRLIST
(0U << 28) | // Reserved
(0U << 29) | // Reserved
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
(0U << 31); // MOVRS
Res.eax = (0U << 0) | // SHA512
(0U << 1) | // SM3
(0U << 2) | // SM4
(0U << 3) | // RAO_INT
(0U << 4) | // AVX_VNNI
(0U << 5) | // AVX512_BF16
(0U << 6) | // LASS (Linear Address Space Separation)
(0U << 7) | // CMPCCXADD
(0U << 8) | // ARCH_PERFMON_EXT
(0U << 9) | // Reserved
(0U << 10) | // FAST_REP_MOVSB
(0U << 11) | // FAST_REP_STOSB
(0U << 12) | // FAST_REP_CMPSB_SCASB
(0U << 13) | // Reserved
(0U << 14) | // Reserved
(0U << 15) | // Reserved
(0U << 16) | // Reserved
(0U << 17) | // FRED (Flexible Return and Event Delivery)
(0U << 18) | // LKGS (Load into Kernel GS Base)
(0U << 19) | // WRMSRNS
(0U << 20) | // NMI_SRC
(0U << 21) | // AMX_FP16
(0U << 22) | // HRESET
(0U << 23) | // AVX_IFMA
(0U << 24) | // Reserved
(0U << 25) | // Reserved
(0U << 26) | // LAM (Linear Address Masking)
(0U << 27) | // MSRLIST
(0U << 28) | // Reserved
(0U << 29) | // Reserved
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
(0U << 31); // MOVRS
// Bits 4-31 currently reserved.
Res.ebx = (0U << 0) | // PPIN
@@ -1109,7 +1091,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
(1 << 23) | // MMX
(1 << 24) | // FXSAVE/FXRSTOR
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
(1 << 26) | // 1 gigabit pages
(0 << 26) | // 1 gigabit pages
(SUPPORTS_RDTSCP << 27) | // RDTSCP
(0 << 28) | // Reserved
(1 << 29) | // Long Mode
@@ -1359,7 +1341,7 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
: CTX {ctx}
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO != 0}
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO}
, GetCPUID {GetCPUID_Syscall} {
Cores = CTX->HostFeatures.CPUMIDRs.size();
+202 -553
View File
@@ -1,10 +1,4 @@
// SPDX-License-Identifier: MIT
#include "Utils/crc32.h"
#include "FEXCore/Utils/LogManager.h"
#include "FEXCore/Utils/MathUtils.h"
#include "FEXCore/Utils/TypeDefines.h"
#include "FEXCore/fextl/memory.h"
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/SpinWaitLock.h>
#include <Interface/Context/Context.h>
@@ -22,14 +16,10 @@
#include <FEXHeaderUtils/Filesystem.h>
#include <algorithm>
#include <git_version.h>
#include <span>
#include <xxhash.h>
#include <FEXCore/Utils/AllocatorHooks.h>
#include <fstream>
namespace FEXCore {
@@ -42,42 +32,6 @@ ExecutableFileInfo::ExecutableFileInfo(fextl::unique_ptr<HLE::SourcecodeMap> Map
#endif
ExecutableFileInfo::~ExecutableFileInfo() = default;
MappedCodeCacheFile::~MappedCodeCacheFile() {
if (CacheManager) {
CacheManager->UnregisterMappedCodeBuffer(*this);
}
#ifndef _WIN32
if (!CodeBuffer.empty()) {
FEXCore::Allocator::munmap(CodeBuffer.data(), CodeBuffer.size_bytes());
}
#elif defined(_M_ARM64EC)
if (!CodeBuffer.empty()) {
FEXCore::Allocator::VirtualFree(CodeBuffer.data(), CodeBuffer.size_bytes());
}
#endif
}
void AbstractCodeCache::RegisterMappedCodeBuffer(MappedCodeCacheFile& Code) {
MappedCodeBuffers.push_back(Code.CodeBuffer);
// Unregister on destruction of Code
Code.CacheManager = this;
}
void AbstractCodeCache::UnregisterMappedCodeBuffer(MappedCodeCacheFile& Code) {
std::erase_if(MappedCodeBuffers, [&](const auto& Elem) { return Elem.data() == Code.CodeBuffer.data(); });
}
bool AbstractCodeCache::IsAddressInMappedCodeBuffer(uintptr_t Address) const {
for (const auto& Range : MappedCodeBuffers) {
auto Start = reinterpret_cast<uintptr_t>(Range.data());
if (Address >= Start && Address < Start + Range.size_bytes()) {
return true;
}
}
return false;
}
fextl::string CodeMap::GetBaseFilename(const ExecutableFileInfo& MainExecutable, bool AddNombSuffix) {
auto FileId = MainExecutable.FileId;
@@ -112,22 +66,16 @@ fextl::map<CodeMapFileId, CodeMap::ParsedContents> CodeMap::ParseCodeMap(std::if
break;
}
Ret[Info.ExternalFileId].Filename = std::move(Filename);
} else if ((Entry.FileId == SetExecutableFileId::Marker32.FileId && Entry.BlockOffset == SetExecutableFileId::Marker32.BlockOffset) ||
(Entry.FileId == SetExecutableFileId::Marker64.FileId && Entry.BlockOffset == SetExecutableFileId::Marker64.BlockOffset)) {
} else if (Entry.FileId == SetExecutableFileId {}.Marker.FileId && Entry.BlockOffset == SetExecutableFileId {}.Marker.BlockOffset) {
CodeMapFileId ExecutableFileId;
File.read(reinterpret_cast<char*>(&ExecutableFileId), sizeof(ExecutableFileId));
if (!File) {
break;
}
Ret[ExecutableFileId].ExecutableBitness =
(Entry.FileId == SetExecutableFileId::Marker32.FileId && Entry.BlockOffset == SetExecutableFileId::Marker32.BlockOffset) ? 32 : 64;
Ret[ExecutableFileId].IsExecutable = true;
} else {
if (!Ret.contains(Entry.FileId)) {
if (Entry.FileId == 0xffff'ffff'ffff'ffff) {
ERROR_AND_DIE_FMT("Malformed code map");
} else {
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
}
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
} else {
Ret[Entry.FileId].Blocks.insert(Entry.BlockOffset);
}
@@ -233,8 +181,8 @@ void CodeMapWriter::AppendLibraryLoad(const FEXCore::ExecutableFileInfo& FileInf
AppendData(std::as_bytes(std::span {Data, TotalSize}));
}
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo, bool Is64Bit) {
CodeMap::SetExecutableFileId Data {Is64Bit ? CodeMap::SetExecutableFileId::Marker64 : CodeMap::SetExecutableFileId::Marker32, FileInfo.FileId};
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo) {
CodeMap::SetExecutableFileId Data {.ExecutableFileId = FileInfo.FileId};
AppendData(std::span {reinterpret_cast<const std::byte*>(&Data), sizeof(Data)});
}
@@ -273,12 +221,19 @@ CodeCache::CodeCache(ContextImpl& CTX_)
: CTX(CTX_) {}
CodeCache::~CodeCache() = default;
uint64_t CodeCache::ComputeCodeMapId(std::string_view Filename, int FD) {
if (Filename.empty()) {
return 0xffff'ffff'ffff'ffff;
}
// For now, we just use the file path as an identifier.
// TODO: Ensure the hash is unique enough to distinguish executables while remaining independent of the installation location
return XXH3_64bits(Filename.data(), Filename.size());
}
struct CodeCacheHeader {
std::array<char, 4> Magic = ExpectedMagic;
// Version history:
// 1: Initial version
// 2: Padding code buffer data to enable direct mapping
uint32_t FormatVersion = 2;
uint32_t FormatVersion = 1;
uint8_t FEXVersion[20] = {};
uint32_t NumBlocks;
uint32_t NumCodePages;
@@ -305,7 +260,7 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
std::ranges::copy(GIT_HASH, header.FEXVersion);
header.NumBlocks = LookupCache.BlockList.size();
header.NumCodePages = LookupCache.CodePages.size();
header.CodeBufferSize = FEXCore::AlignUp(CodeBuffer->AllocatedSpaceUsed(), Utils::FEX_PAGE_SIZE);
header.CodeBufferSize = CTX.LatestOffset;
header.NumRelocations = Relocations.size();
header.SerializedBaseAddress = SerializedBaseAddress;
::write(fd, &header, sizeof(header));
@@ -328,7 +283,7 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
Guest -= SourceBinary.FileStartVA;
::write(fd, &Guest, sizeof(Guest));
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->GetBufferBase());
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->Ptr);
::write(fd, &HostCode, sizeof(HostCode));
uint64_t NumCodePages = Host->CodePages.size();
::write(fd, &NumCodePages, sizeof(NumCodePages));
@@ -352,19 +307,12 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
}
// Dump the host code (relocated for position-independent serialization)
std::span CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->GetBufferBase()),
reinterpret_cast<std::byte*>(CodeBuffer->GetBufferBase()) + CodeBuffer->AllocatedSpaceUsed());
std::span CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->Ptr), reinterpret_cast<std::byte*>(CodeBuffer->Ptr) + CTX.LatestOffset);
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, true)) {
LOGMAN_THROW_A_FMT(false, "Failed to apply code relocations");
return false;
}
::write(fd, CodeBufferData.data(), CodeBufferData.size());
// Pad to next page in file for mmap
{
auto PaddedSize = AlignUp(lseek(fd, 0, SEEK_CUR), Utils::FEX_PAGE_SIZE);
::ftruncate(fd, PaddedSize);
lseek(fd, PaddedSize, SEEK_SET);
}
// Dump code pages
static_assert(OrderedContainer<decltype(LookupCache.CodePages)>, "Non-deterministic data source");
@@ -382,6 +330,167 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
return true;
}
bool CodeCache::LoadData(Core::InternalThreadState* Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& BinarySection) {
if (!EnableCodeCaching) {
return true;
}
namespace ranges = std::ranges;
// Read file header
CodeCacheHeader header {};
::memcpy(&header, MappedCacheFile, sizeof(header));
MappedCacheFile += sizeof(header);
LogMan::Msg::IFmt("Cache load: {:5} blocks; base={:#14x}; off={:#9x}-{:#09x}; {:016x} {}", header.NumBlocks, BinarySection.FileStartVA,
BinarySection.BeginVA - BinarySection.FileStartVA, BinarySection.EndVA - BinarySection.FileStartVA,
BinarySection.FileInfo.FileId, BinarySection.FileInfo.Filename);
if (!ranges::equal(header.Magic, header.ExpectedMagic)) {
LogMan::Msg::EFmt("Invalid cache file header");
return false;
}
if (!ranges::equal(header.FEXVersion, GIT_HASH)) {
LogMan::Msg::IFmt("Cache generated from old FEX version {:02x}, current is {:02x}; skipping", fmt::join(header.FEXVersion, ""),
fmt::join(GIT_HASH, ""));
return false;
}
if (header.NumBlocks == 0) {
// Valid caches are never empty
LogMan::Msg::IFmt("Code cache empty, aborting");
return false;
}
// Read guest<->host block mappings
using BlockListEntry = decltype(GuestToHostMap::BlockList)::value_type;
fextl::vector<BlockListEntry> BlockList(header.NumBlocks);
{
for (auto& BlockPtr : BlockList) {
::memcpy(&BlockPtr.first, MappedCacheFile, sizeof(BlockPtr.first));
MappedCacheFile += sizeof(BlockPtr.first);
::memcpy(&BlockPtr.second.HostCode, MappedCacheFile, sizeof(BlockPtr.second.HostCode));
MappedCacheFile += sizeof(BlockPtr.second.HostCode);
uint64_t NumGuestPages;
::memcpy(&NumGuestPages, MappedCacheFile, sizeof(NumGuestPages));
MappedCacheFile += sizeof(NumGuestPages);
BlockPtr.second.CodePages.resize(NumGuestPages);
::memcpy(BlockPtr.second.CodePages.data(), MappedCacheFile, std::span {BlockPtr.second.CodePages}.size_bytes());
MappedCacheFile += std::span {BlockPtr.second.CodePages}.size_bytes();
}
// Constrain BlockList to the given ExecutableFileSectionInfo
LOGMAN_THROW_A_FMT(ranges::is_sorted(BlockList, [](auto& a, auto& b) { return a.first < b.first; }), "Expected sorted block list");
auto begin = ranges::lower_bound(BlockList, BinarySection.BeginVA - BinarySection.FileStartVA, std::less {}, &BlockListEntry::first);
auto end =
ranges::upper_bound(begin, BlockList.end(), BinarySection.EndVA - BinarySection.FileStartVA - 1, std::less {}, &BlockListEntry::first);
if (begin == end) {
// Not an error since there is just no data to load
LogMan::Msg::IFmt("No blocks cached in this range, aborting");
return true;
}
BlockList.erase(end, BlockList.end());
BlockList.erase(BlockList.begin(), begin);
}
// Read relocations
fextl::vector<FEXCore::CPU::Relocation> Relocations(header.NumRelocations, FEXCore::CPU::Relocation::Default());
::memcpy(Relocations.data(), MappedCacheFile, Relocations.size() * sizeof(Relocations[0]));
MappedCacheFile += Relocations.size() * sizeof(Relocations[0]);
// Pad to next page in file, which contains CodeBuffer data
MappedCacheFile = reinterpret_cast<std::byte*>(AlignUp(reinterpret_cast<uintptr_t>(MappedCacheFile), Utils::FEX_PAGE_SIZE));
// Prepare CodeBuffer: Page aligned and big enough to hold all cached data
auto Lock = std::unique_lock {CTX.CodeBufferWriteMutex};
if (Thread) {
if (auto Prev = Thread->CPUBackend->CheckCodeBufferUpdate()) {
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
auto lk = Thread->LookupCache->AcquireWriteLock();
Thread->LookupCache->ChangeGuestToHostMapping(*Prev, *CTX.GetLatest()->LookupCache, lk);
}
}
auto CodeBuffer = CTX.GetLatest();
LOGMAN_THROW_A_FMT(reinterpret_cast<uintptr_t>(CodeBuffer->Ptr) % 0x1000 == 0, "Expected CodeBuffer base to be page-aligned");
const auto Delta = AlignUp(CTX.LatestOffset, 0x1000) - CTX.LatestOffset;
CTX.LatestOffset += Delta;
while (CTX.LatestOffset + header.CodeBufferSize > CodeBuffer->UsableSize()) {
if (Thread) {
CTX.ClearCodeCache(Thread);
CodeBuffer = CTX.GetLatest();
LogMan::Msg::IFmt("Increased code buffer size to {} MiB for cache load", CodeBuffer->AllocatedSize / 1024 / 1024);
} else {
ERROR_AND_DIE_FMT("Cannot extend codebuffer without thread!");
}
}
// Read CodeBuffer data from file. Make sure the destination is page-aligned.
// TODO: Only load the data needed for the selected section
auto CodeBufferRange =
std::as_writable_bytes(std::span {CodeBuffer->Ptr, CodeBuffer->UsableSize()}).subspan(CTX.LatestOffset, header.CodeBufferSize);
::memcpy(CodeBufferRange.data(), MappedCacheFile, header.CodeBufferSize);
MappedCacheFile += header.CodeBufferSize;
CTX.LatestOffset += header.CodeBufferSize;
// Apply FEX relocations
auto Ret = ApplyCodeRelocations(BinarySection.FileStartVA, CodeBufferRange, Relocations, false);
LOGMAN_THROW_A_FMT(Ret == true, "Failed to apply code cache relocations");
{
auto& LookupCache = *CodeBuffer->LookupCache;
auto WriteLock = LookupCache.AcquireWriteLock();
// Register blocks to LookupCache
for (auto& [Guest, Host] : BlockList) {
for (auto& CodePage : Host.CodePages) {
CodePage += BinarySection.FileStartVA;
}
auto HostCode = reinterpret_cast<void*>(Host.HostCode + reinterpret_cast<uintptr_t>(CodeBufferRange.data()));
LookupCache.AddBlockMapping(Guest + BinarySection.FileStartVA, std::move(Host.CodePages), HostCode, WriteLock);
}
// Register loaded code ranges
fextl::vector<uint64_t> Entrypoints;
for (uint32_t i = 0; i < header.NumCodePages; ++i) {
uint64_t CodePage;
memcpy(&CodePage, MappedCacheFile, sizeof(CodePage));
CodePage += BinarySection.FileStartVA;
MappedCacheFile += sizeof(CodePage);
uint64_t NumEntrypoints;
memcpy(&NumEntrypoints, MappedCacheFile, sizeof(NumEntrypoints));
MappedCacheFile += sizeof(NumEntrypoints);
Entrypoints.resize(NumEntrypoints);
memcpy(Entrypoints.data(), MappedCacheFile, NumEntrypoints * sizeof(Entrypoints[0]));
MappedCacheFile += NumEntrypoints * sizeof(Entrypoints[0]);
for (auto& Entrypoint : Entrypoints) {
Entrypoint += BinarySection.FileStartVA;
}
if (LookupCache.AddBlockExecutableRange(Entrypoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE, WriteLock)) {
CTX.SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
}
if (EnableCodeCacheValidation) {
fextl::set<uint64_t> GuestBlocks, HostBlocks;
for (auto& [Guest, Host] : BlockList) {
GuestBlocks.insert(Guest + BinarySection.FileStartVA);
HostBlocks.insert(Host.HostCode);
}
Validate(BinarySection, std::move(GuestBlocks), HostBlocks, CodeBufferRange);
}
return true;
}
void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<uint64_t> GuestBlocks, const fextl::set<uint64_t>& HostBlocks,
std::span<std::byte> CachedCode) {
LOGMAN_THROW_A_FMT(!HostBlocks.empty(), "Tried to validate without any host blocks");
@@ -397,7 +506,7 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
ERROR_AND_DIE_FMT("Failed to create cache load validation context");
}
ValidationThread.reset(ValidationCTX->CreateThread(nullptr));
ValidationThread.reset(ValidationCTX->CreateThread(0, 0, nullptr));
auto Frame = ValidationThread->CurrentFrame;
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_GDT] = &ValidationGDT[0];
@@ -418,12 +527,11 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
while (CachedCode.size_bytes() > NewCodeBuffer->UsableSize()) {
ValidationCTX->ClearCodeCache(ValidationThread.get());
NewCodeBuffer = ValidationCTX->GetLatest();
LogMan::Msg::IFmt("Increased cache validation code buffer size to {} MiB", NewCodeBuffer->TotalAllocationSize() / 1024 / 1024);
LogMan::Msg::IFmt("Increased cache validation code buffer size to {} MiB", NewCodeBuffer->AllocatedSize / 1024 / 1024);
}
std::span<std::byte> CodeBufferRangeRef =
std::as_writable_bytes(std::span {NewCodeBuffer->GetBufferBase(), NewCodeBuffer->GetBufferBase() + NewCodeBuffer->UsableSize()})
.subspan(0, CachedCode.size_bytes());
std::as_writable_bytes(std::span {NewCodeBuffer->Ptr, NewCodeBuffer->Ptr + NewCodeBuffer->UsableSize()}).subspan(0, CachedCode.size_bytes());
while (!GuestBlocks.empty()) {
auto [CompiledBlocks, _, _2, _3, _4] = ValidationCTX->CompileCode(ValidationThread.get(), *GuestBlocks.begin(), 0 /* TODO: Set MaxInst? */);
@@ -439,10 +547,10 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
}));
(void)ApplyCodeRelocations(Section.FileStartVA, CodeBufferRangeRef, NewRelocations, false);
if (NewCodeBuffer->AllocatedSpaceUsed() <= CodeBufferRangeRef.size()) {
if (ValidationCTX->LatestOffset <= CodeBufferRangeRef.size()) {
// Reference compilation produced fewer bytes than our cache, so validation is going to fail.
// Make sure we don't output any garbage bytes though.
CodeBufferRangeRef = CodeBufferRangeRef.subspan(0, NewCodeBuffer->AllocatedSpaceUsed());
CodeBufferRangeRef = CodeBufferRangeRef.subspan(0, ValidationCTX->LatestOffset);
}
auto [Mismatch, _] = std::mismatch(CodeBufferRangeRef.begin(), CodeBufferRangeRef.end(), CachedCode.begin());
@@ -469,7 +577,7 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
if (tail->RIP >= Section.BeginVA && tail->RIP < Section.EndVA) {
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, _] =
ValidationCTX->GenerateIR(ValidationThread.get(), tail->RIP, false, FEXCore::Config::Get_MAXINST());
fextl::ostringstream ss;
fextl::stringstream ss;
FEXCore::IR::Dump(&ss, &*IRView);
LogMan::Msg::EFmt("IR:\n{}", ss.str());
} else {
@@ -493,126 +601,9 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
// Reset Context state for next validation
ValidationThread->LookupCache->ClearCache(ValidationThread->LookupCache->AcquireWriteLock());
NewCodeBuffer->Reset();
ValidationCTX->LatestOffset = 0;
LogMan::Msg::IFmt(" successfully validated cache");
}
static inline void ApplySymbolLiteralRelocation(ContextImpl& CTX, const CPU::RelocNamedSymbolLiteral::NamedSymbol Symbol,
uint64_t GuestEntry, CPU::Arm64Emitter& Emitter, bool ForStorage) {
// Generate a literal so we can place it
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Symbol);
Emitter.dc64(Pointer);
}
static inline bool
ApplyThunkMoveRelocation(ContextImpl& CTX, const IR::SHA256Sum* Symbol, uint32_t RegisterIndex, CPU::Arm64Emitter& Emitter, bool ForStorage) {
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(*Symbol));
if (Pointer == ~0ULL) {
return false;
}
// TODO: Pointers are required to fit within 48-bit VA space.
// But forcing 6-byte broke relocations.
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
return true;
}
static inline void ApplyRIPLiteralRelocation(ContextImpl& CTX, uint64_t GuestRIP, uint64_t GuestEntry, CPU::Arm64Emitter& Emitter) {
Emitter.dc64(GuestEntry + GuestRIP);
}
static inline void
ApplyRIPMoveRelocation(ContextImpl& CTX, uint64_t GuestRIP, uint8_t RegisterIndex, uint64_t GuestEntry, CPU::Arm64Emitter& Emitter) {
uint64_t Pointer = GuestRIP + GuestEntry;
// TODO: Pointers are required to fit within 48-bit VA space.
// But forcing 6-byte broke relocations.
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
}
static inline int64_t ReadLiveGuestData(uint64_t SiteAddress, uint8_t ValueSize) {
uint64_t Raw = 0;
memcpy(&Raw, reinterpret_cast<const void*>(SiteAddress), ValueSize);
// manual sign-extension from guest live bytes
if (ValueSize == 1) {
return (int8_t)Raw;
} else if (ValueSize == 2) {
return (int16_t)Raw;
} else if (ValueSize == 4) {
return (int32_t)Raw;
} else {
return (int64_t)Raw;
}
}
static inline void ApplyPatchableDataRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), ReadLiveGuestData(SiteAddress, ValueSize),
CPU::Arm64Emitter::PadType::DOPAD);
}
static inline void ApplyPatchableRIPLiteralRelocation(uint64_t SiteAddress, uint8_t ValueSize, CPU::Arm64Emitter& Emitter) {
Emitter.dc64(SiteAddress + ValueSize + ReadLiveGuestData(SiteAddress, ValueSize));
}
static inline void ApplyPatchableRIPMoveRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
const uint64_t Target = SiteAddress + ValueSize + ReadLiveGuestData(SiteAddress, ValueSize);
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Target, CPU::Arm64Emitter::PadType::DOPAD);
}
static inline void ApplyPatchableCRCMoveRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
const uint64_t Target = FEXCore::Utils::crc32(reinterpret_cast<const uint8_t*>(SiteAddress), ValueSize);
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Target, CPU::Arm64Emitter::PadType::DOPAD);
}
bool CodeCache::ApplyPackedCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
std::span<const DiskCache::BlobSmallRelocation> SmallRelocs,
std::span<const DiskCache::BlobThunkRelocation> ThunkRelocs) {
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
for (auto& Reloc : SmallRelocs) {
LOGMAN_THROW_A_FMT(Reloc.Offset < Code.size_bytes(), "Invalid relocation offset");
Emitter.SetCursorOffset(Reloc.Offset);
switch ((CPU::RelocationTypes)Reloc.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
ApplySymbolLiteralRelocation(CTX, (CPU::RelocNamedSymbolLiteral::NamedSymbol)Reloc.Named.Symbol, GuestEntry, Emitter, false);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
ApplyRIPLiteralRelocation(CTX, Reloc.RIPLiteral.GuestRIP, GuestEntry, Emitter);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
ApplyRIPMoveRelocation(CTX, Reloc.RIPMove.GuestRIP, Reloc.RIPMove.RegisterIndex, GuestEntry, Emitter);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE: {
ApplyPatchableDataRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
Reloc.PatchableData.RegisterIndex, Emitter);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
ApplyPatchableRIPLiteralRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize, Emitter);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE: {
ApplyPatchableRIPMoveRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
Reloc.PatchableData.RegisterIndex, Emitter);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_CRC_MOVE: {
ApplyPatchableCRCMoveRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
Reloc.PatchableData.RegisterIndex, Emitter);
break;
}
default: ERROR_AND_DIE_FMT("Unknown packed relocation type {}", ToUnderlying((CPU::RelocationTypes)Reloc.Type));
}
}
for (auto& Reloc : ThunkRelocs) {
LOGMAN_THROW_A_FMT(Reloc.Offset < Code.size_bytes(), "Invalid relocation offset");
Emitter.SetCursorOffset(Reloc.Offset);
if (!ApplyThunkMoveRelocation(CTX, (const IR::SHA256Sum*)Reloc.SymbolHash, Reloc.RegisterIndex, Emitter, false)) {
return false;
}
}
return true;
LogMan::Msg::IFmt("\tSuccessfully validated cache");
}
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
@@ -620,26 +611,35 @@ bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> C
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
LOGMAN_THROW_A_FMT(Reloc.Header.Offset < Code.size_bytes(), "Invalid relocation offset");
Emitter.SetCursorOffset(Reloc.Header.Offset);
switch (Reloc.Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
ApplySymbolLiteralRelocation(CTX, Reloc.NamedSymbolLiteral.Symbol, GuestEntry, Emitter, ForStorage);
// Generate a literal so we can place it
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Reloc.NamedSymbolLiteral.Symbol);
Emitter.dc64(Pointer);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
if (!ApplyThunkMoveRelocation(CTX, &Reloc.NamedThunkMove.Symbol, Reloc.NamedThunkMove.RegisterIndex, Emitter, ForStorage)) {
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// TODO: Pointers are required to fit within 48-bit VA space.
// But forcing 6-byte broke relocations.
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer,
CPU::Arm64Emitter::PadType::DOPAD);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
ApplyRIPLiteralRelocation(CTX, Reloc.GuestRIP.GuestRIP, GuestEntry, Emitter);
Emitter.dc64(GuestEntry + Reloc.GuestRIP.GuestRIP);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
ApplyRIPMoveRelocation(CTX, Reloc.GuestRIP.GuestRIP, Reloc.GuestRIP.RegisterIndex, GuestEntry, Emitter);
uint64_t Pointer = Reloc.GuestRIP.GuestRIP + GuestEntry;
// TODO: Pointers are required to fit within 48-bit VA space.
// But forcing 6-byte broke relocations.
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIP.RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
break;
}
@@ -650,355 +650,4 @@ bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> C
return true;
}
fextl::unique_ptr<MappedCodeCacheFile>
CodeCache::LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo& FileInfo, uint64_t FileStartVA) {
if (!EnableCodeCaching) {
return nullptr;
}
FEXCORE_PROFILE_SCOPED("LoadCache");
// Read file header
CodeCacheHeader header {};
::memcpy(&header, CacheFile.data(), sizeof(header));
if (!std::ranges::equal(header.Magic, header.ExpectedMagic)) {
LogMan::Msg::EFmt("Invalid cache file header");
return nullptr;
}
if (!std::ranges::equal(header.FEXVersion, GIT_HASH)) {
LogMan::Msg::IFmt("Cache generated from old FEX version {:02x}, current is {:02x}; skipping", fmt::join(header.FEXVersion, ""),
fmt::join(GIT_HASH, ""));
return nullptr;
}
if (header.NumBlocks == 0) {
// Valid caches are never empty
LogMan::Msg::IFmt("Code cache empty, aborting");
return nullptr;
}
// Skip over BlockEntry data since it won't be used until EnableLoadedSection
// TODO: Store direct offset to relocations in the header
auto* BlockListStart = CacheFile.data() + sizeof(header);
auto* Cursor = BlockListStart;
for (uint32_t i = 0; i < header.NumBlocks; ++i) {
Cursor += sizeof(uint64_t); // guest address
Cursor += sizeof(uint64_t); // host code address
uint64_t NumGuestCodePages;
::memcpy(&NumGuestCodePages, Cursor, sizeof(NumGuestCodePages));
Cursor += sizeof(NumGuestCodePages);
Cursor += NumGuestCodePages * sizeof(uint64_t);
}
auto Relocations = std::span {reinterpret_cast<const FEXCore::CPU::Relocation*>(Cursor), header.NumRelocations};
Cursor += Relocations.size_bytes();
// Pad to next page to get the code buffer data
Cursor = reinterpret_cast<std::byte*>(AlignUp(reinterpret_cast<uintptr_t>(Cursor), Utils::FEX_PAGE_SIZE));
auto CodeDataInFile = std::span {Cursor, header.CodeBufferSize};
#ifndef _WIN32
// Allocate target memory for post-relocation code. This is PROT_NONE until
// the first execution, so that contents can be lazily populated in a
// frontend-provided segfault handler.
void* CodeBufferAllocation = Allocator::mmap(nullptr, header.CodeBufferSize, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
if (CodeBufferAllocation == MAP_FAILED) {
LogMan::Msg::EFmt("Failed to reserve target memory for code cache");
return nullptr;
}
auto CodeBuffer = std::span {static_cast<std::byte*>(CodeBufferAllocation), header.CodeBufferSize};
#elif defined(_M_ARM64EC)
// TODO: Implement lazy mapping on Windows
// NOTE: The executed code must have MEM_EXTENDED_PARAMETER_EC_CODE set, so we can't operate on the mapped cache file directly
void* CodeBufferAllocation = Allocator::VirtualAlloc(header.CodeBufferSize, true);
if (!CodeBufferAllocation) {
LogMan::Msg::EFmt("Failed to allocate code cache memory");
return nullptr;
}
auto CodeBuffer = std::span {reinterpret_cast<std::byte*>(CodeBufferAllocation), header.CodeBufferSize};
#else // WoW64
// TODO: Implement lazy mapping on Windows
auto CodeBuffer = CodeDataInFile;
#endif
// Group relocations by page
size_t NumPages = header.CodeBufferSize / Utils::FEX_PAGE_SIZE;
fextl::vector<MappedCodeCacheFile::PageRelocationRange> PageRelocationRanges(NumPages, {0, 0});
auto RelocBaseOffset = std::as_bytes(Relocations).data() - CacheFile.data();
auto RelocIt = Relocations.begin();
for (size_t Page = 0; Page < NumPages; ++Page) {
auto EndRelocIt = std::upper_bound(RelocIt, Relocations.end(), Page,
[](auto& Page, auto& Reloc) { return Page < Reloc.Header.Offset / Utils::FEX_PAGE_SIZE; });
PageRelocationRanges.at(Page) = {static_cast<uint32_t>(RelocBaseOffset + (RelocIt - Relocations.begin()) * sizeof(CPU::Relocation)),
static_cast<uint32_t>(EndRelocIt - RelocIt)};
RelocIt = EndRelocIt;
}
auto Storage = FEXCore::Allocator::aligned_alloc(alignof(MappedCodeCacheFile), sizeof(MappedCodeCacheFile));
return fextl::unique_ptr<MappedCodeCacheFile>(
new (Storage) MappedCodeCacheFile {this, CacheFile, CodeDataInFile, CodeBuffer, BlockListStart, header.NumBlocks, header.NumCodePages,
std::move(PageRelocationRanges), fextl::vector<bool>(NumPages), FileStartVA});
}
bool CodeCache::EnableLoadedSection(Core::InternalThreadState* Thread, MappedCodeCacheFile& Code, const ExecutableFileSectionInfo& BinarySection) {
if (!EnableCodeCaching) {
return true;
}
namespace ranges = std::ranges;
FEXCORE_PROFILE_SCOPED("EnableLoadedSection");
// Read block list from cache file
// TODO: Store section-ized BlockLists in cache file
using BlockListEntry = decltype(GuestToHostMap::BlockList)::value_type;
fextl::vector<BlockListEntry> BlockList(Code.NumBlocks);
{
auto* Cursor = Code.BlockListInFile;
for (auto& BlockPtr : BlockList) {
::memcpy(&BlockPtr.first, Cursor, sizeof(BlockPtr.first));
Cursor += sizeof(BlockPtr.first);
::memcpy(&BlockPtr.second.HostCode, Cursor, sizeof(BlockPtr.second.HostCode));
Cursor += sizeof(BlockPtr.second.HostCode);
uint64_t NumGuestPages;
::memcpy(&NumGuestPages, Cursor, sizeof(NumGuestPages));
Cursor += sizeof(NumGuestPages);
BlockPtr.second.CodePages.resize(NumGuestPages);
::memcpy(BlockPtr.second.CodePages.data(), Cursor, std::span {BlockPtr.second.CodePages}.size_bytes());
Cursor += std::span {BlockPtr.second.CodePages}.size_bytes();
}
// Constrain BlockList to the given ExecutableFileSectionInfo
LOGMAN_THROW_A_FMT(ranges::is_sorted(BlockList, [](auto& a, auto& b) { return a.first < b.first; }), "Expected sorted block list");
auto begin = ranges::lower_bound(BlockList, BinarySection.BeginVA - BinarySection.FileStartVA, std::less {}, &BlockListEntry::first);
auto end =
ranges::upper_bound(begin, BlockList.end(), BinarySection.EndVA - BinarySection.FileStartVA - 1, std::less {}, &BlockListEntry::first);
if (begin == end) {
LogMan::Msg::IFmt("No blocks cached in this range, aborting");
return true;
}
BlockList.erase(end, BlockList.end());
BlockList.erase(BlockList.begin(), begin);
}
LogMan::Msg::IFmt("Cache load: {:5} blocks; base={:#14x}; off={:#9x}-{:#09x}; {:016x} {}", BlockList.size(), BinarySection.FileStartVA,
BinarySection.BeginVA - BinarySection.FileStartVA, BinarySection.EndVA - BinarySection.FileStartVA,
BinarySection.FileInfo.FileId, BinarySection.FileInfo.Filename);
if (EnableLazyCodeCaching) {
LogMan::Msg::IFmt(" lazy mapping: base={:#14x} -> host={}; cache_source={}", BinarySection.FileStartVA,
fmt::ptr(Code.CodeBuffer.data()), fmt::ptr(Code.MappedFile.data()));
}
// Register blocks to LookupCache.
// The host addresses will point into the protected code buffer, so that FEX
// can lazily apply relocations on first execution of each page.
auto CodeBuffer = CTX.GetLatest();
{
FEXCORE_PROFILE_SCOPED("Decode");
auto& LookupCache = *CodeBuffer->LookupCache;
auto WriteLock = LookupCache.AcquireWriteLock();
for (auto& [Guest, Block] : BlockList) {
for (auto& CodePage : Block.CodePages) {
CodePage += BinarySection.FileStartVA;
}
LOGMAN_THROW_A_FMT(Block.HostCode < Code.CodeBuffer.size_bytes(), "Host offset {:#x} out of range ({:#x})", Block.HostCode,
Code.CodeBuffer.size_bytes());
auto HostCode = &Code.CodeBuffer[Block.HostCode];
LookupCache.AddBlockMapping(Guest + BinarySection.FileStartVA, std::move(Block.CodePages), HostCode, WriteLock);
}
// Guest code pages
auto* Cursor = Code.CodeBufferInFile.data() + Code.CodeBufferInFile.size_bytes();
fextl::vector<uint64_t> Entrypoints;
for (uint32_t i = 0; i < Code.NumCodePages; ++i) {
uint64_t CodePage;
memcpy(&CodePage, Cursor, sizeof(CodePage));
CodePage += BinarySection.FileStartVA;
Cursor += sizeof(CodePage);
uint64_t NumEntrypoints;
memcpy(&NumEntrypoints, Cursor, sizeof(NumEntrypoints));
Cursor += sizeof(NumEntrypoints);
Entrypoints.resize(NumEntrypoints);
memcpy(Entrypoints.data(), Cursor, std::span {Entrypoints}.size_bytes());
Cursor += std::span {Entrypoints}.size_bytes();
for (auto& Entrypoint : Entrypoints) {
Entrypoint += BinarySection.FileStartVA;
}
if (LookupCache.AddBlockExecutableRange(Entrypoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE, WriteLock)) {
CTX.SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
}
#ifndef _WIN32
if (!EnableLazyCodeCaching || EnableCodeCacheValidation) {
#else
// TODO: Implement lazy mapping on Windows
if (true) {
#endif
auto Range = SelectCodeRangeToFinalize(Code, 0, Code.CodeBuffer.size_bytes() / Utils::FEX_PAGE_SIZE);
FinalizeCodePages(Code, Range);
}
if (EnableCodeCacheValidation) {
fextl::set<uint64_t> GuestBlocks, HostBlocks;
for (auto& [Guest, Host] : BlockList) {
GuestBlocks.insert(Guest + BinarySection.FileStartVA);
HostBlocks.insert(Host.HostCode);
}
Validate(BinarySection, std::move(GuestBlocks), HostBlocks, Code.CodeBuffer);
}
return true;
}
} // namespace FEXCore::Context
namespace FEXCore {
static std::span<CPU::Relocation> SpanPageRelocations(const MappedCodeCacheFile& Code, size_t PageIndex) {
auto [Offset, Count] = Code.PageRelocationRanges.at(PageIndex);
return std::span {reinterpret_cast<FEXCore::CPU::Relocation*>(Code.MappedFile.data() + Offset), Count};
}
std::span<std::byte> AbstractCodeCache::SelectCodeRangeToFinalize(MappedCodeCacheFile& Code, size_t StartPage, size_t EndPage) {
// First, check if we were racing another thread in loading this range
if (std::find(Code.LoadedPages.begin() + StartPage, Code.LoadedPages.begin() + EndPage, false) == Code.LoadedPages.begin() + EndPage) {
return {};
}
LOGMAN_THROW_A_FMT(StartPage < EndPage, "Invalid page range [{}, {})", StartPage, EndPage);
LOGMAN_THROW_A_FMT(EndPage <= Code.NumPages(), "End page {} out of range ({})", EndPage, Code.NumPages());
// Include any pages that have relocations or block link records crossing
// into the current page range. This ensures we don't attempt to finalize
// any page twice, partially apply FEX relocations, or trigger page loads
// during block linking.
while (EndPage < Code.NumPages()) {
auto PageRelocs = SpanPageRelocations(Code, EndPage - 1);
if (!PageRelocs.empty()) {
auto It = std::prev(PageRelocs.end());
size_t RelocEnd = It->Header.Offset + 16 /* Upper bound for relocation size */;
if (RelocEnd > EndPage * Utils::FEX_PAGE_SIZE) {
++EndPage;
continue;
}
}
// Check for trailing block link
{
auto PageRelocs = SpanPageRelocations(Code, EndPage);
if (!PageRelocs.empty() && PageRelocs.begin()->Header.Offset < EndPage * Utils::FEX_PAGE_SIZE + 0x18) {
++EndPage;
continue;
}
}
break;
};
while (StartPage != 0) {
auto PageRelocs = SpanPageRelocations(Code, StartPage - 1);
if (!PageRelocs.empty()) {
auto It = std::prev(PageRelocs.end());
size_t RelocEnd = It->Header.Offset + 16 /* Upper bound for relocation size */;
if (RelocEnd > StartPage * Utils::FEX_PAGE_SIZE) {
--StartPage;
continue;
}
}
// Check for trailing block link
{
auto PageRelocs = SpanPageRelocations(Code, StartPage);
if (!PageRelocs.empty() && PageRelocs.begin()->Header.Offset < StartPage * Utils::FEX_PAGE_SIZE + 0x18) {
--StartPage;
continue;
}
}
break;
};
return Code.CodeBuffer.subspan(StartPage * Utils::FEX_PAGE_SIZE, (EndPage - StartPage) * Utils::FEX_PAGE_SIZE);
}
} // namespace FEXCore
namespace FEXCore::Context {
void CodeCache::FinalizeCodePages(MappedCodeCacheFile& Code, std::span<std::byte> CodeRange) {
const size_t StartOffset = CodeRange.data() - Code.CodeBuffer.data();
const auto StartPage = StartOffset / Utils::FEX_PAGE_SIZE;
const auto EndPage = StartPage + CodeRange.size_bytes() / Utils::FEX_PAGE_SIZE;
const size_t Size = CodeRange.size_bytes();
// None of the selected pages should be loaded at all; otherwise, SelectCodeRangeToFinalize returned inconsistent ranges
LOGMAN_THROW_A_FMT(std::find(Code.LoadedPages.begin() + StartPage, Code.LoadedPages.begin() + EndPage, true) == Code.LoadedPages.begin() + EndPage,
"Inconsistent page load state");
FEXCORE_PROFILE_SCOPED("FinalizeCodePages");
#ifndef _WIN32
// Atomicity is critical when making the finalized code data visible.
// We ensure this by remapping a temporary buffer onto the PROT_NONE
// placeholder page in CodeBuffer. Some constraints to keep in mind are:
// 1. Pages can't be write-only (readability is implicitly added), so
// we can't change CodeBuffer from PROT_NONE to PROT_WRITE even for just
// a short duration
// 2. Naive mremap from CodeBufferInFile to CodeBuffer would leave a gap in
// the former, which would make cleanup overly complicated
//
// Due to (1), we can't apply relocations in place (CodeBufferInFile); at
// least a secondary buffer is needed for execution (CodeBuffer).
// Due to (2), a third buffer is temporarily allocated here and freed on
// completion. The final code data is computed here and then the memory
// is remapped onto CodeBuffer.
auto* Staging = reinterpret_cast<std::byte*>(Allocator::VirtualAlloc(nullptr, Size, true));
if (!Staging) {
ERROR_AND_DIE_FMT("Failed to allocate {} bytes of staging memory for code-cache finalization", Size);
}
// Copy code from the cache file to the staging buffer
memcpy(Staging, Code.CodeBufferInFile.data() + StartOffset, Size);
// Apply relocations
auto StagingSpan = std::span {Staging, Size};
for (size_t i = StartPage; i < EndPage; ++i) {
auto PageRelocations = SpanPageRelocations(Code, i);
(void)ApplyCodeRelocations(Code.GuestBase, StagingSpan, PageRelocations, false);
Code.LoadedPages[i] = true;
}
// Atomically make the finalized code data visible by remapping the staging
// buffer onto the requested CodeBuffer window. MREMAP_DONTUNMAP is used to
// leave the old VA range reserved so that we can cleanly deallocate it
// through Allocator.
void* RemapResult = ::mremap(Staging, Size, Size, MREMAP_FIXED | MREMAP_MAYMOVE | MREMAP_DONTUNMAP, CodeRange.data());
if (RemapResult == MAP_FAILED) {
ERROR_AND_DIE_FMT("{}: mremap failed: {}", __FUNCTION__, errno);
}
Allocator::VirtualFree(Staging, Size);
// Release resident file pages that will no longer be needed. The VA range is left allocated to allow cleanup with a single VirtualFree.
Allocator::VirtualDontNeed(Code.CodeBufferInFile.data() + StartOffset, Size);
#else
// TODO: Implement lazy mapping on Windows
#ifdef _M_ARM64EC
memcpy(Code.CodeBuffer.data() + StartOffset, Code.CodeBufferInFile.data() + StartOffset, Size);
#endif
for (size_t i = StartPage; i < EndPage; ++i) {
auto PageRelocations = SpanPageRelocations(Code, i);
(void)ApplyCodeRelocations(Code.GuestBase, Code.CodeBuffer, PageRelocations, false);
Code.LoadedPages[i] = true;
}
#endif
ARMEmitter::Emitter::ClearICache(CodeRange.data(), Size);
}
} // namespace FEXCore::Context
+58 -162
View File
@@ -30,7 +30,6 @@ $end_info$
#include "Interface/IR/RegisterAllocationData.h"
#include "Utils/Allocator.h"
#include "Utils/Allocator/HostAllocator.h"
#include "Utils/crc32.h"
#include <FEXCore/Utils/SpinWaitLock.h>
#include "Utils/variable_length_integer.h"
@@ -90,7 +89,7 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
if (Config.BlockJITNaming() || Config.GlobalJITNaming() || Config.LibraryJITNaming()) {
// Only initialize symbols file if enabled. Ensures we don't pollute /tmp with empty files.
Symbols.InitFile(Features.ProcessPID);
Symbols.InitFile();
}
uint64_t FrequencyCounter = FEXCore::GetCycleCounterFrequency();
@@ -104,16 +103,6 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
// Track atomic TSO emulation configuration.
UpdateAtomicTSOEmulationConfig();
#ifndef _WIN32
// Check if the kernel supports getrandom().
uint64_t Probe {};
HostRNGAvailable = FHU::Syscalls::getrandom(&Probe, sizeof(Probe), 0) == sizeof(Probe);
#else
HostRNGAvailable = true;
#endif
DiskCache.Init(this);
}
struct GetFrameBlockInfoResult {
@@ -353,32 +342,22 @@ void ContextImpl::SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState
}
bool ContextImpl::InitCore() {
if (CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHINGWIP()) {
// Start with a larger code buffer to avoid resizes that would discard code
StartMaximalCodeBuffer();
}
// Initialize the CPU core signal handlers & DispatcherConfig
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
// Set up the SignalDelegator config since core is initialized.
SignalDelegation->SetConfig(Dispatcher->MakeSignalDelegatorConfig());
#if defined(_WIN32) && !defined(ARCHITECTURE_arm64ec)
// WOW64 always needs the interrupt fault check to be enabled.
Config.NeedsPendingInterruptFaultCheck = true;
#endif
if (Config.GdbServer) {
// If gdbserver is enabled then this needs to be enabled.
Config.NeedsPendingInterruptFaultCheck = true;
}
if constexpr (BLOCK_DEBUGGING) {
// If the developer wants to do any single-stepping points or watch points.
// Add them here.
//
// eg:
// BlockDebuggerTracker.AllTargetSingleStep();
// BlockDebuggerTracker.AddSingleStepTarget(0x14000'0000ULL);
// BlockDebuggerTracker.AddWriteWatchPoint(0x420BA5ED);
}
return true;
}
@@ -397,11 +376,11 @@ void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
}
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this, Thread);
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>(this);
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
Thread->CurrentFrame->State.L1Pointer = Thread->LookupCache->GetL1Pointer();
Thread->CurrentFrame->State.L1Mask = Thread->LookupCache->GetScaledL1PointerMask();
@@ -410,20 +389,28 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
Dispatcher->InitThreadPointers(Thread);
Thread->PassManager->AddDefaultPasses(this);
Thread->PassManager->AddDefaultValidationPasses();
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
// Create CPU backend
Thread->PassManager->InsertRegisterAllocationPass(this);
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
// We finalize *after* the CPU backend is initialized, as the CPU backend will
// provide necessary register information to the register allocation pass.
Thread->PassManager->Finalize();
}
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(const FEXCore::Core::CPUState* NewThreadState) {
FEXCore::Core::InternalThreadState*
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState) {
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
.CTX = this,
};
FEXCore::Allocator::VirtualName("FEXMem_ThreadState", Thread, sizeof(*Thread));
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
Thread->CurrentFrame->State.rip = InitialRIP;
// Copy over the new thread state to the new object
if (NewThreadState) {
memcpy(&Thread->CurrentFrame->State, NewThreadState, sizeof(FEXCore::Core::CPUState));
@@ -469,6 +456,7 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
if (Config.StrictInProcessSplitLocks) {
FEXCore::Utils::SpinWaitLock::unlock(&StrictSplitLockMutex);
}
return;
}
}
@@ -483,7 +471,7 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
void ContextImpl::OnCodeBufferAllocated(const fextl::shared_ptr<CPU::CodeBuffer>& Buffer) {
if (Config.GlobalJITNaming()) {
Symbols.RegisterJITSpace(Buffer->GetBufferBase(), Buffer->TotalAllocationSize());
Symbols.RegisterJITSpace(Buffer->Ptr, Buffer->AllocatedSize);
}
{
@@ -508,11 +496,11 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, boo
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
FEXCore::File::File FD = FEXCore::File::File::GetStdERR();
fextl::ostringstream out;
fextl::stringstream out;
auto NewIR = IREmitter->ViewIR();
FEXCore::IR::Dump(&out, &NewIR);
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", NewIR.PostRA() ? "post" : "pre", GuestRIP, out.str());
}
};
bool ContextImpl::CheckIfBlockIsCacheable(FEXCore::Core::InternalThreadState& Thread, uint64_t GuestRIP, uint64_t MaxInst) {
return Thread.FrontendDecoder->CheckIfCacheable(Thread, reinterpret_cast<const uint8_t*>(GuestRIP), GuestRIP, MaxInst);
@@ -529,8 +517,6 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
bool HasCustomIR {};
bool WantsDiskCachePatching = DiskCache.IsReadingDiskCache() || DiskCache.IsWritingDiskCache();
if (HasCustomIRHandlers.load(std::memory_order_relaxed)) {
std::shared_lock lk(CustomIRMutex);
auto Handler = CustomIRHandlers.find(GuestRIP);
@@ -543,14 +529,18 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
}
if (!HasCustomIR) {
const auto* GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
const uint8_t* GuestCode {};
GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
Thread->FrontendDecoder->DecodeLoop(GuestCode);
bool HadDispatchError {false};
bool HadInvalidInst {false};
const auto* BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
const auto& CodeBlocks = BlockInfo->Blocks;
Thread->FrontendDecoder->DecodeInstructionsAtEntry(Thread, GuestCode, GuestRIP, MaxInst);
Thread->OpDispatcher->BeginFunction(GuestRIP, &CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
auto CodeBlocks = &BlockInfo->Blocks;
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
AreMonoHacksActive() && MonoBackpatcherBlock.load(std::memory_order_relaxed) == GuestRIP);
const auto GPRSize = Thread->OpDispatcher->GetGPROpSize();
@@ -564,17 +554,11 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
}
#endif
for (size_t j = 0; j < CodeBlocks.size(); ++j) {
const auto& Block = CodeBlocks[j];
// Dispatch failures and invalid instructions terminate only the decoded
// block that contains them. Other block targets in the same multiblock
// compilation unit are independent entry paths.
bool HadDispatchError {false};
bool HadInvalidInst {false};
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
#ifdef ZYDIS_DISASSEMBLER
if (FEXCore::Config::Get_X86DISASSEMBLE() && CodeBlocks.size() > 1) {
if (FEXCore::Config::Get_X86DISASSEMBLE() && CodeBlocks->size() > 1) {
LogMan::Msg::IFmt(" Block {} Entry={:#x} NumInsts={}", j, Block.Entry, Block.NumInstructions);
}
#endif
@@ -582,7 +566,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
bool BlockInForceTSOValidRange = false;
auto InstForceTSOIt = ForceTSOInstructions.end();
if (ForceTSOValidRanges.Contains({Block.Entry, Block.Entry + Block.Size})) {
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); It != ForceTSOInstructions.end() && *It < Block.Entry + Block.Size) {
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); *It < Block.Entry + Block.Size) {
InstForceTSOIt = It;
BlockInForceTSOValidRange = true;
}
@@ -591,16 +575,18 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
// Set the block entry point
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
uint64_t BlockInstructionsLength {};
// Reset any block-specific state
Thread->OpDispatcher->StartNewBlock();
const uint64_t InstsInBlock = Block.NumInstructions;
uint64_t InstsInBlock = Block.NumInstructions;
if (InstsInBlock == 0) {
// Special case for an empty instruction block.
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry - GuestRIP));
}
uint64_t BlockInstructionsLength {};
for (size_t i = 0; i < InstsInBlock; ++i) {
uint64_t InstAddress = Block.Entry + BlockInstructionsLength;
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
@@ -644,12 +630,9 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL || Block.ForceFullSMCDetection) {
auto ExistingCodePtr = reinterpret_cast<uint8_t*>(Block.Entry + BlockInstructionsLength);
auto InstAddressReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
auto Value = FEXCore::Utils::crc32(ExistingCodePtr, DecodedInfo->InstSize);
auto CRC = WantsDiskCachePatching ?
Thread->OpDispatcher->_PatchableGuestCRC(IR::OpSize::i64Bit, Value, (int64_t)ExistingCodePtr, DecodedInfo->InstSize) :
Thread->OpDispatcher->Constant(Value);
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CRC, InstAddressReg, DecodedInfo->InstSize);
std::array<uint8_t, 0x10> CodeOriginal;
memcpy(CodeOriginal.data(), ExistingCodePtr, DecodedInfo->InstSize);
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CodeOriginal, InstAddressReg, DecodedInfo->InstSize);
auto InvalidateCodeCond = Thread->OpDispatcher->CondJump(CodeChanged);
@@ -658,19 +641,13 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
Thread->OpDispatcher->SetTrueJumpTarget(InvalidateCodeCond, CodeWasChangedBlock);
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
Thread->OpDispatcher->StartNewBlock();
// Generate a relocatable entry for invalidation purposes.
auto EntryToInvalidate = Thread->OpDispatcher->_EntrypointOffset(GPRSize, 0);
auto NewRIP = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
// Invalidate and exit the function
Thread->OpDispatcher->_ThreadRemoveCodeEntry(EntryToInvalidate, NewRIP);
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
Thread->OpDispatcher->SetFalseJumpTarget(InvalidateCodeCond, NextOpBlock);
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
Thread->OpDispatcher->StartNewBlock();
}
if (TableInfo && TableInfo->OpcodeDispatcher.OpDispatch) {
@@ -718,8 +695,6 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::INVALID_INST ||
Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::BAD_RELOCATION) {
Thread->OpDispatcher->InvalidOp(DecodedInfo);
} else if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::UNIMPLEMENTED_INST) {
Thread->OpDispatcher->UnimplementedOp(DecodedInfo);
} else {
Thread->OpDispatcher->NoExecOp(DecodedInfo);
}
@@ -759,9 +734,9 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
#endif
Thread->OpDispatcher->Finalize();
}
Thread->FrontendDecoder->DelayedDisownBuffer();
Thread->FrontendDecoder->DelayedDisownBuffer();
}
IR::IREmitter* IREmitter = Thread->OpDispatcher.get();
@@ -802,8 +777,6 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, NeedsAddGuestCodeRanges] =
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
if (!IRView) {
Thread->FrontendDecoder->ValidateDisownedOrFree();
Thread->OpDispatcher->ValidateDisownedOrFree();
// OpDispatcher IR already released in this case.
return {{}, nullptr, 0, 0, false};
}
@@ -817,8 +790,6 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
// Raced to compile, release the OpDispatcher IR.
Thread->OpDispatcher->DelayedDisownBuffer();
Thread->FrontendDecoder->ValidateDisownedOrFree();
Thread->OpDispatcher->ValidateDisownedOrFree();
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
.DebugData = nullptr,
.StartAddr = 0,
@@ -837,8 +808,6 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
// Release the IR
Thread->OpDispatcher->DelayedDisownBuffer();
Thread->FrontendDecoder->ValidateDisownedOrFree();
Thread->OpDispatcher->ValidateDisownedOrFree();
return {
.CompiledCode = std::move(CompiledCode),
.DebugData = std::move(DebugData),
@@ -849,17 +818,6 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
}
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst) {
if constexpr (BLOCK_DEBUGGING) {
// Block debugging logic is hand-written and needs to be handled with care.
// Force MaxInst to only be one in this case.
MaxInst = 1;
// If the entrypoint is part of the single step targets then single step it.
if (BlockDebuggerTracker.IsSingleStepTarget(GuestRIP)) {
return CompileSingleStep(Frame, GuestRIP);
}
}
auto Thread = Frame->Thread;
FEXCORE_PROFILE_SCOPED("CompileBlock");
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedJITTime);
@@ -875,47 +833,6 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
return HostCode;
}
Thread->FrontendDecoder->SetupDecodeInstructionsAtEntry(Thread, GuestRIP, MaxInst);
std::optional<ExecutableFileSectionInfo> Region = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
std::optional<DiskCache::CodeHitData> Hit;
std::optional<uint64_t> DiskCacheGuestCodeKey;
{
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedDiskCacheLookupTime);
Hit = DiskCache.Lookup(Thread, Region, GuestRIP, DiskCacheGuestCodeKey);
if (Hit && !DiskCache.IsValidating()) {
auto LoadedCode = Thread->CPUBackend->LoadCachedCode(Hit->HostCode);
if (LoadedCode.BlockBegin) {
for (auto& CodePage : Hit->GuestPages) {
if (Thread->LookupCache->AddBlockExecutableRange(Thread, Hit->EntryPointRIPs, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
LOGMAN_THROW_A_FMT(Hit->EntryPointRIPs.size() == Hit->EntryPointHostOffsets.size(), "Mismatched Disk Cache entrypoint pairs!");
uintptr_t CachedHostCode = 0;
for (size_t i = 0; i < Hit->EntryPointRIPs.size(); i++) {
void* HostAddr = LoadedCode.BlockBegin + Hit->EntryPointHostOffsets[i];
Thread->LookupCache->AddBlockMapping(Thread, Hit->EntryPointRIPs[i], Hit->GuestPages, HostAddr);
if (Hit->EntryPointRIPs[i] == GuestRIP) {
CachedHostCode = reinterpret_cast<uintptr_t>(HostAddr);
}
}
LOGMAN_THROW_A_FMT(CachedHostCode != 0, "Couldn't find GuestRIP in Disk Cache entrypoints!");
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedDiskCacheHitCount, 1);
Thread->FrontendDecoder->DelayedDisownBuffer();
Thread->FrontendDecoder->ValidateDisownedOrFree();
Thread->OpDispatcher->ValidateDisownedOrFree();
return CachedHostCode;
}
}
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedDiskCacheMissCount, 1);
}
// Accumulate a JIT count now, as even if another thread raced us, it should count as a compile.
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedJITCount, 1);
@@ -928,13 +845,6 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
return reinterpret_cast<uintptr_t>(CodePtr);
}
if (DiskCacheGuestCodeKey && Hit && DiskCache.IsValidating()) {
DiskCache.Validate(*DiskCacheGuestCodeKey, *Hit, CompiledCode, Region);
}
// if this ever fires, we need to serialize the offset into disk cache
LOGMAN_THROW_A_FMT(StartAddr == GuestRIP, "StartAddr offset from GuestRIP");
// The core managed to compile the code.
if (Config.BlockJITNaming()) {
auto FragmentBasePtr = CompiledCode.BlockBegin;
@@ -974,6 +884,11 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
}
}
// Clear any relocations that might have been generated
if (!CodeCache.IsGeneratingCache) {
Thread->CPUBackend->ClearRelocations();
}
fextl::vector<uint64_t> CodePages;
if (NeedsAddGuestCodeRanges) {
@@ -989,37 +904,19 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
}
}
// Disk Cache
if (!CodeCache.IsGeneratingCache) {
if (DiskCacheGuestCodeKey) {
std::span<const FEXCore::CPU::Relocation> Relocations;
if (DebugData && DebugData->Relocations) {
Relocations = *DebugData->Relocations;
}
std::span<const uint8_t> GuestCode = {reinterpret_cast<const uint8_t*>(StartAddr), Length};
const Frontend::Decoder::DecodedBlockInformation* BlockInfo =
NeedsAddGuestCodeRanges ? Thread->FrontendDecoder->GetDecodedBlockInfo() : nullptr;
DiskCache.Store(Thread, Region, GuestRIP, *DiskCacheGuestCodeKey, GuestCode, CompiledCode, Relocations, BlockInfo);
}
if (CodeMapWriter && Region && Region->FileStartVA != 0) {
CodeMapWriter->AppendBlock(*Region, GuestRIP);
}
}
// Insert to lookup cache
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
Thread->LookupCache->AddBlockMapping(Thread, GuestAddr, CodePages, HostAddr);
}
// Clear any relocations that might have been generated
if (!CodeCache.IsGeneratingCache) {
Thread->CPUBackend->ClearRelocations();
if (CodeMapWriter) {
auto Region = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
if (Region && Region->FileStartVA != 0) {
CodeMapWriter->AppendBlock(*Region, GuestRIP);
}
}
Thread->FrontendDecoder->ValidateDisownedOrFree();
Thread->OpDispatcher->ValidateDisownedOrFree();
return (uintptr_t)CodePtr;
}
@@ -1032,7 +929,6 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
Thread->FrontendDecoder->SetupDecodeInstructionsAtEntry(Thread, GuestRIP, 1);
auto [CompiledCode, DebugData, StartAddr, Length, _] = CompileCode(Thread, GuestRIP, 1);
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
if (CodePtr == nullptr) {
File diff suppressed because it is too large. Load diff
-258
View File
@@ -1,258 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Interface/Core/JIT/Relocations.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/CPUBackend.h"
#include <FEXCore/Core/CodeCache.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/File.h>
#include <FEXCore/Utils/WorkQueueThread.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_set.h>
#include <FEXCore/fextl/robin_map.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/vector.h>
#include <stdint.h>
#include <mutex>
#include <optional>
#include <span>
#include <xxhash.h>
namespace FEXCore {
namespace Context {
class ContextImpl;
}
namespace DiskCache {
namespace MesaFOZ {
#define FOSSILIZE_BLOB_HASH_LENGTH 40 /* SHA1 hexadecimal string length */
struct __attribute__((packed)) foz_payload_key {
uint8_t bytes[FOSSILIZE_BLOB_HASH_LENGTH];
};
struct __attribute__((packed)) foz_payload_header {
uint32_t payload_size;
uint32_t format;
uint32_t crc;
uint32_t uncompressed_size;
};
struct mesa_index_db_file_entry;
} // namespace MesaFOZ
class IndexedDB;
struct MemoryLRUKey {
uint64_t LookupKey;
XXH128_hash_t GuestHash;
uint64_t GuestFootprint;
uint32_t Size;
};
struct IndexEntry {
IndexedDB* DB;
uint64_t Offset;
uint32_t Size;
uint32_t GuestSize;
XXH128_hash_t GuestHash;
fextl::shared_ptr<fextl::vector<uint8_t>> MemoryBlob;
fextl::vector<uint32_t> GuestExtents;
std::optional<fextl::list<MemoryLRUKey>::iterator> LRUEntry;
};
struct IndexCacheHead {
struct IndexEntry MainEntry;
uint64_t MainEntryFootprint;
fextl::unique_ptr<fextl::multimap<uint64_t, IndexEntry>> MoreEntries; // sorted by guest footprint
};
struct __attribute__((packed)) BlobFixedHeader {
uint32_t GuestSize;
uint32_t HostSize;
uint32_t EntryPointCount;
uint32_t SmallRelocCount;
uint32_t ThunkRelocCount;
XXH128_hash_t GuestHash;
};
// packed struct for types 0, 2 and 3. type 1 is bigger and separate below
struct __attribute__((packed)) BlobSmallRelocation {
uint32_t Offset;
uint8_t Type;
union {
struct __attribute__((packed)) {
uint32_t Symbol;
} Named;
struct __attribute__((packed)) {
uint64_t GuestRIP;
} RIPLiteral;
struct __attribute__((packed)) {
uint8_t RegisterIndex;
uint64_t GuestRIP;
} RIPMove;
struct __attribute__((packed)) {
uint8_t RegisterIndex;
uint8_t ValueSize;
uint32_t SiteOffset;
} PatchableData;
};
};
// type 1, implicit
struct __attribute__((packed)) BlobThunkRelocation {
uint32_t Offset;
uint8_t RegisterIndex;
uint8_t SymbolHash[32]; // sha256sum in the real RelocNamedThunkMove
};
struct CodeHitData {
fextl::vector<uint8_t> Blob;
std::span<uint8_t> HostCode;
std::span<const uint64_t> GuestPages;
std::span<uint64_t> EntryPointRIPs;
std::span<const uint32_t> EntryPointHostOffsets;
// the spans above point to memory owned by the Blob vec, so it's important this can't be copied
CodeHitData() = default;
CodeHitData(CodeHitData&&) = default;
CodeHitData& operator=(CodeHitData&&) = default;
CodeHitData(const CodeHitData&) = delete;
CodeHitData& operator=(const CodeHitData&) = delete;
};
using Index = fextl::robin_map<uint64_t, IndexCacheHead>;
class FOZFile {
public:
bool Open(const fextl::string& CacheFileName, bool ReadOnly);
bool Lock(uint32_t TimeoutMS) {
if (!FD) {
return false;
}
return FD->Lock(TimeoutMS);
}
bool Unlock() {
if (!FD) {
return false;
}
return FD->Unlock();
}
File::File::FileHandleType GetHandle() {
return FD ? FD->GetHandle() : (File::File::FileHandleType)-1;
}
ssize_t Size();
bool ReadAll(fextl::vector<uint8_t>& Out); // from first blob
bool ReadBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
bool WriteBlob(const MesaFOZ::foz_payload_key& Key, std::span<const std::span<const uint8_t>> BlobChunks, uint64_t& OutBlobOffset);
private:
static constexpr uint32_t OPEN_LOCK_TIMEOUT_MS = 100;
fextl::string FileName;
fextl::unique_ptr<File::File> FD;
bool ReadOnly = false;
};
class IndexedDB {
public:
bool Open(const fextl::string& CacheDBName, bool ReadOnly);
void PopulateIndex(Index& CacheIndex, bool& FoundMetadata);
bool ReadCacheBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
bool StoreCacheBlob(const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey, std::span<const uint8_t> Blob,
MesaFOZ::mesa_index_db_file_entry& IndexEntry, std::span<const uint8_t> IndexBlob);
bool Full() const {
return MaxSizeReached;
}
private:
// stores run on the Writer, so returning quick isn't as important
static constexpr uint32_t STORE_LOCK_TIMEOUT_MS = 1000;
static constexpr uint64_t BIG_MAPPING_SIZE = 1ULL << 33;
FOZFile CacheFOZ;
uint8_t* CacheFileMapping = nullptr;
std::atomic<uint64_t> CacheFileSize;
FOZFile IndexFOZ;
bool ReadOnly = false;
bool MaxSizeReached = false;
FEX_CONFIG_OPT(MaxFileSize, DISKCACHEMAXFILESIZE);
};
class DiskCache {
public:
void Init(FEXCore::Context::ContextImpl* CTX);
std::optional<CodeHitData> Lookup(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP,
std::optional<uint64_t>& GuestCodeKey);
void Validate(uint64_t GuestCodeKey, const CodeHitData& Hit, const CPU::CPUBackend::CompiledCode& CompiledCode,
std::optional<ExecutableFileSectionInfo> Region);
bool Store(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP, uint64_t GuestCodeKey,
std::span<const uint8_t> GuestCode, const CPU::CPUBackend::CompiledCode& CompiledCode,
std::span<const FEXCore::CPU::Relocation> Relocations, const Frontend::Decoder::DecodedBlockInformation* DecodedBlockInfo);
bool IsWritingDiskCache() const {
return WritingDiskCache;
}
bool IsReadingDiskCache() const {
return ReadingDiskCache;
}
bool IsValidating() const {
return Validation;
}
private:
bool OpenCacheDB(const fextl::string& CacheDBName, bool ReadOnly);
uint64_t MakeLookupKey(Core::InternalThreadState* Thread, const uint64_t ModuleOffset, bool Writable, bool MonoBackpatcher);
IndexEntry* LookupLocked(const uint64_t LookupKey, const XXH128_hash_t& GuestHash, const uint64_t GuestFootprint);
bool ReadingDiskCache {};
bool WritingDiskCache {};
FEXCore::Context::ContextImpl* CTX;
XXH128_hash_t BucketHash;
fextl::vector<fextl::unique_ptr<IndexedDB>> ROCacheDBs;
fextl::unique_ptr<IndexedDB> RWCacheDB;
Index Index;
std::mutex IndexLock;
bool FoundMetadata = false;
struct CacheStoreWorkItem;
struct PruneMemoryLRUWorkItem;
std::atomic<uint64_t> MemoryLRUCurrentSize {};
std::mutex MemoryLRULock;
fextl::list<MemoryLRUKey> MemoryLRU;
// the Writer holds references to all this stuff above and needs to be last
fextl::unique_ptr<WorkQueueThread> Writer;
FEX_CONFIG_OPT(EnableDiskCache, DISKCACHE);
FEX_CONFIG_OPT(Validation, DISKCACHEVALIDATION);
FEX_CONFIG_OPT(MapDiskCacheFiles, DISKCACHEFILEMAPPING);
FEX_CONFIG_OPT(RelocationFilter, DISKCACHERELOCATIONFILTER);
FEX_CONFIG_OPT(AnonCaching, DISKCACHEANONCACHING);
FEX_CONFIG_OPT(BasePathOverride, DISKCACHEPATH);
FEX_CONFIG_OPT(RODBNames, DISKCACHERODBNAMES);
FEX_CONFIG_OPT(MemoryLRUMaxSize, DISKCACHEMEMORYSIZE);
uint64_t MemoryLRUEvictThreshold = MemoryLRUMaxSize / 25;
};
static constexpr uint16_t AnonPrefixGuestBytes = 64;
// The current version of the diskcache.
// This must be changed any time codegen changes occur!
// Be aware of the impact of changing this frequently!
static constexpr uint16_t FormatVersion = 29;
static constexpr uint32_t LOOKUP_KEY_MAX_BUCKET_DEPTH = 500;
} // namespace DiskCache
} // namespace FEXCore
@@ -121,7 +121,7 @@ void Dispatcher::EmitDispatcher() {
ldr(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
FillSpecialRegs(TMP1, TMP2, {.SetFIZ = false, .SetPredRegs = true});
FillSpecialRegs(TMP1, TMP2, false, true);
// As ARM64EC uses this as an entrypoint for both guest calls and host returns, opportunistically try to return
// using the call-ret stack to avoid unbalancing it.
@@ -153,8 +153,9 @@ void Dispatcher::EmitDispatcher() {
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 18);
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
(void)tbz(TMP1, 0, &l_NotECCode);
@@ -276,15 +277,8 @@ void Dispatcher::EmitDispatcher() {
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
// Trigger segfault if any deferred signals are pending
constexpr size_t InterruptPageOffset =
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState);
if constexpr (InterruptPageOffset <= 32760) {
str(ARMEmitter::XReg::zr, STATE, InterruptPageOffset);
} else {
// Need to use vector 128-bit store for this range.
// Doesn't matter which register we use to store.
str(ARMEmitter::QReg::q0, STATE, InterruptPageOffset);
}
strb(ARMEmitter::XReg::zr, STATE,
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
#endif
};
@@ -364,7 +358,7 @@ void Dispatcher::EmitDispatcher() {
ldr(ARMEmitter::XReg::x4, &l_CompileSingleStep);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t>(ARMEmitter::Reg::r4);
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
} else {
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP }
}
@@ -514,64 +508,6 @@ void Dispatcher::EmitDispatcher() {
(void)b(&LoopTop);
}
{
// All dynamic and static registers are spilled coming in to this handler.
// It's also the end of block and RIP might have changed, so we jump directly to the top of the loop.
ThreadDispatchSyscallHandler = GetCursorAddress<uint64_t>();
// Store in the state that we are in a syscall
// 16bit LoadConstant to be a single instruction
// This gives the signal handler a value to check to see if we are in a syscall at all
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, 0xFFFF);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerObj));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerFunc));
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
} else {
blr(ARMEmitter::Reg::r3);
}
// Fix the stack and any values that were stepped on
// Syscall result is in any static register that the frontend desired.
FillStaticRegs({
.OptionalReg = ARMEmitter::Reg::r1,
.OptionalReg2 = ARMEmitter::Reg::r2,
});
// Now the registers we've spilled are back in their original host registers
// We can safely claim we are no longer in a syscall
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
// Now go back to the regular dispatcher loop
(void)b(&LoopTop);
}
{
// All dynamic and static registers are spilled coming in to this handler.
// It's also the end of block and RIP might have changed, so we jump directly to the top of the loop.
ThreadDispatchRemoveCodeEntry = GetCursorAddress<uint64_t>();
// Arguments are already in x0, x1. Just jump to the handler.
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadRemoveCodeEntryFromJIT));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
blr(ARMEmitter::Reg::r2);
}
FillStaticRegs({
.OptionalReg = ARMEmitter::Reg::r1,
.OptionalReg2 = ARMEmitter::Reg::r2,
});
// Now go back to the regular dispatcher loop
(void)b(&LoopTop);
}
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
auto Address = GetCursorAddress<uint64_t>();
@@ -616,12 +552,7 @@ void Dispatcher::EmitDispatcher() {
EmitF64F2XM1();
EmitF64Scale();
EmitF64Atan();
F64Log2Constants Log2C;
EmitF64FYL2X(Log2C);
EmitF64FYL2XP1(Log2C);
EmitF64Log2Constants(Log2C);
EmitF64FPREM();
EmitF64FPREM1();
EmitF64FYL2X();
// Interpreter fallbacks
{
@@ -1674,13 +1605,18 @@ void Dispatcher::EmitF64Atan() {
// JIT-inlined double-precision y * log2(x) for the F64 reduced precision x87 path.
// Input: VTMP1 = x, VTMP2 = y. Output: VTMP1 = y * log2(x).
// Constants in `C` are shared with EmitF64FYL2XP1 and emitted by EmitF64Log2Constants.
void Dispatcher::EmitF64FYL2X(F64Log2Constants& C) {
// Algorithm: atanh-based log via s = f/(2+f) with 9-term polynomial, scaled by 1/ln(2),
// then multiplied by y.
void Dispatcher::EmitF64FYL2X() {
F64FYL2XHandlerAddress = GetCursorAddress<uint64_t>();
constexpr auto Accum = ARMEmitter::VReg::v2;
ARMEmitter::ForwardLabel Fallback;
ARMEmitter::ForwardLabel NoNorm;
ARMEmitter::ForwardLabel Sqrt2Label, Log2eLabel;
ARMEmitter::ForwardLabel BiasLabel;
ARMEmitter::ForwardLabel P0Label, P1Label, P2Label, P3Label, P4Label, P5Label, P6Label, P7Label, P8Label;
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
@@ -1697,58 +1633,63 @@ void Dispatcher::EmitF64FYL2X(F64Log2Constants& C) {
(void)b(ARMEmitter::Condition::CC_EQ, &Fallback);
fmov(ARMEmitter::Size::i64Bit, TMP3, VTMP2.D());
// k = unbiased exponent; m bits = mantissa | (0x3FF << 52).
// Extract k and normalize mantissa m into [1.0, 2.0).
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1023);
ubfx(ARMEmitter::Size::i64Bit, TMP1, TMP1, 0, 52);
movz(ARMEmitter::Size::i64Bit, TMP4, 0x3FF0, 48);
ldr(TMP4, &BiasLabel);
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4);
// Index = top 6 mantissa bits (bits 51..46 of m).
ubfx(ARMEmitter::Size::i64Bit, TMP4, TMP1, 46, 6);
// Set m as F64 in VTMP1.
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
// Load (recip, logc) from LUT[index].
(void)adr(TMP1, &C.Table);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
ldp<ARMEmitter::IndexType::OFFSET>(VTMP2.D(), Accum.D(), TMP1, 0);
// If m > sqrt(2), halve m and increment k.
ldr(VTMP2.D(), &Sqrt2Label);
fcmp(VTMP1.D(), VTMP2.D());
(void)b(ARMEmitter::Condition::CC_LE, &NoNorm);
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP2, 0.5f);
fmul(VTMP1.D(), VTMP1.D(), VTMP2.D());
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
(void)Bind(&NoNorm);
// Stash logc bits in TMP4 so Accum can be reused for the polynomial.
fmov(ARMEmitter::Size::i64Bit, TMP4, Accum.D());
// r = recip * m - 1.
fmul(VTMP1.D(), VTMP2.D(), VTMP1.D());
ldr(VTMP2.D(), &C.One);
// f = m - 1; s = f / (2 + f); TMP1 stashes s, VTMP1 holds s^2.
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP2, 1.0f);
fsub(VTMP1.D(), VTMP1.D(), VTMP2.D());
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP2, 2.0f);
fadd(VTMP2.D(), VTMP1.D(), VTMP2.D());
fdiv(Accum.D(), VTMP1.D(), VTMP2.D());
fmul(VTMP1.D(), Accum.D(), Accum.D());
fmov(ARMEmitter::Size::i64Bit, TMP1, Accum.D());
// Horner: poly = a0 + r*(a1 + r*(a2 + ... + r*a7)).
ldr(Accum.D(), &C.A7);
ldr(VTMP2.D(), &C.A6);
// 9-term Horner from 1/19 down to 1/3.
ldr(Accum.D(), &P8Label);
ldr(VTMP2.D(), &P7Label);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A5);
ldr(VTMP2.D(), &P6Label);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A4);
ldr(VTMP2.D(), &P5Label);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A3);
ldr(VTMP2.D(), &P4Label);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A2);
ldr(VTMP2.D(), &P3Label);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A1);
ldr(VTMP2.D(), &P2Label);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A0);
ldr(VTMP2.D(), &P1Label);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &P0Label);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
// log2(1+r) = r * Accum.
fmul(VTMP1.D(), VTMP1.D(), Accum.D());
// ln(1+f) = 2 * s * (1 + s^2 * P(s^2)).
fmul(Accum.D(), VTMP1.D(), Accum.D());
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP2, 1.0f);
fadd(Accum.D(), Accum.D(), VTMP2.D());
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP1);
fmul(Accum.D(), VTMP2.D(), Accum.D());
fadd(Accum.D(), Accum.D(), Accum.D());
// log2(x) = log2(1+r) + log2(center[i]) + k.
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP4);
fadd(VTMP1.D(), VTMP1.D(), VTMP2.D());
// log2(x) = k + ln(1+f)/ln(2); multiply by y.
ldr(VTMP1.D(), &Log2eLabel);
fmul(VTMP1.D(), Accum.D(), VTMP1.D());
scvtf(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP2);
fadd(VTMP1.D(), VTMP1.D(), VTMP2.D());
// result = y * log2(x).
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
fmul(VTMP1.D(), VTMP1.D(), VTMP2.D());
@@ -1769,483 +1710,33 @@ void Dispatcher::EmitF64FYL2X(F64Log2Constants& C) {
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
ret();
}
// JIT-inlined double-precision y * log2(1 + x) for the F64 reduced precision x87 path.
// Input: VTMP1 = x, VTMP2 = y. Output: VTMP1 = y * log2(1 + x).
// Computes v = 1 + x in F64 and runs the same LUT-based log2 as F64FYL2X.
// Loses 1-2 ulps of precision near x=0 (FYL2XP1's original purpose) but
// matches main's lowering and avoids the range-check cliff into a fallback.
void Dispatcher::EmitF64FYL2XP1(F64Log2Constants& C) {
F64FYL2XP1HandlerAddress = GetCursorAddress<uint64_t>();
constexpr auto Accum = ARMEmitter::VReg::v2;
ARMEmitter::ForwardLabel Fallback;
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
// v = 1 + x in Accum so VTMP1/VTMP2 still hold the original x/y at the fallback.
ldr(Accum.D(), &C.One);
fadd(Accum.D(), VTMP1.D(), Accum.D());
// Reject v <= 0, subnormal, NaN, Inf via v's bits in TMP1.
fmov(ARMEmitter::Size::i64Bit, TMP1, Accum.D());
(void)tbnz(TMP1, 63, &Fallback);
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP1, 52);
(void)cbz(ARMEmitter::Size::i64Bit, TMP2, &Fallback);
cmp(ARMEmitter::Size::i64Bit, TMP2, 0x7FF);
(void)b(ARMEmitter::Condition::CC_EQ, &Fallback);
fmov(ARMEmitter::Size::i64Bit, TMP3, VTMP2.D());
// k = unbiased exponent; m bits = mantissa | (0x3FF << 52).
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1023);
ubfx(ARMEmitter::Size::i64Bit, TMP1, TMP1, 0, 52);
movz(ARMEmitter::Size::i64Bit, TMP4, 0x3FF0, 48);
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4);
// Index = top 6 mantissa bits (bits 51..46 of m).
ubfx(ARMEmitter::Size::i64Bit, TMP4, TMP1, 46, 6);
// Set m as F64 in VTMP1.
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
// Load (recip, logc) from LUT[index].
(void)adr(TMP1, &C.Table);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
ldp<ARMEmitter::IndexType::OFFSET>(VTMP2.D(), Accum.D(), TMP1, 0);
// Stash logc bits in TMP4 so Accum can be reused for the polynomial.
fmov(ARMEmitter::Size::i64Bit, TMP4, Accum.D());
// r = recip * m - 1.
fmul(VTMP1.D(), VTMP2.D(), VTMP1.D());
ldr(VTMP2.D(), &C.One);
fsub(VTMP1.D(), VTMP1.D(), VTMP2.D());
// Horner: poly = a0 + r*(a1 + r*(a2 + ... + r*a7)).
ldr(Accum.D(), &C.A7);
ldr(VTMP2.D(), &C.A6);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A5);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A4);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A3);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A2);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A1);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
ldr(VTMP2.D(), &C.A0);
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
// log2(1+r) = r * Accum.
fmul(VTMP1.D(), VTMP1.D(), Accum.D());
// log2(v) = log2(1+r) + log2(center[i]) + k.
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP4);
fadd(VTMP1.D(), VTMP1.D(), VTMP2.D());
scvtf(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP2);
fadd(VTMP1.D(), VTMP1.D(), VTMP2.D());
// result = y * log2(v).
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
fmul(VTMP1.D(), VTMP1.D(), VTMP2.D());
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
ret();
// Fallback path: VTMP1/VTMP2 still hold the original x/y.
(void)Bind(&Fallback);
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FYL2XP1].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FYL2XP1].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
ret();
}
// Shared constants pool for the LUT-based F64 log2 path used by FYL2X and FYL2XP1.
// Emitted once after both handlers; their forward `ldr`/`adr` references are
// patched here. Layout: One (1.0), 8 Horner coefficients A0..A7, then a 64-entry
// LUT of (1/center[i], log2(center[i])) pairs where center[i] = 1 + (i+0.5)/64.
void Dispatcher::EmitF64Log2Constants(F64Log2Constants& C) {
// Constant pool: bias=1.0, sqrt(2), log2(e)=1/ln(2), P8..P0 = 1/19, 1/17, 1/15, ..., 1/3.
Align(16);
(void)Bind(&C.One);
dc64(0x3FF0000000000000ULL); // 1.0
(void)Bind(&C.A0);
dc64(0x3FF71547652B82FEULL); // log2(e) * 1/1
(void)Bind(&C.A1);
dc64(0xBFE71547652B82FEULL); // log2(e) * -1/2
(void)Bind(&C.A2);
dc64(0x3FDEC709DC3A03FDULL); // log2(e) * 1/3
(void)Bind(&C.A3);
dc64(0xBFD71547652B82FEULL); // log2(e) * -1/4
(void)Bind(&C.A4);
dc64(0x3FD2776C50EF9BFEULL); // log2(e) * 1/5
(void)Bind(&C.A5);
dc64(0xBFCEC709DC3A03FDULL); // log2(e) * -1/6
(void)Bind(&C.A6);
dc64(0x3FCA61762A7ADED9ULL); // log2(e) * 1/7
(void)Bind(&C.A7);
dc64(0xBFC71547652B82FEULL); // log2(e) * -1/8
Align(16);
(void)Bind(&C.Table);
dc64(0x3FEFC07F01FC07F0ULL);
dc64(0x3F86FE50B6EF0851ULL); // i= 0
dc64(0x3FEF44659E4A4271ULL);
dc64(0x3FA11CD1D5133413ULL); // i= 1
dc64(0x3FEECC07B301ECC0ULL);
dc64(0x3FAC4DFAB90AAB5FULL); // i= 2
dc64(0x3FEE573AC901E574ULL);
dc64(0x3FB3AA2FDD27F1C3ULL); // i= 3
dc64(0x3FEDE5D6E3F8868AULL);
dc64(0x3FB918A16E46335BULL); // i= 4
dc64(0x3FED77B654B82C34ULL);
dc64(0x3FBE72EC117FA5B2ULL); // i= 5
dc64(0x3FED0CB58F6EC074ULL);
dc64(0x3FC1DCD197552B7BULL); // i= 6
dc64(0x3FECA4B3055EE191ULL);
dc64(0x3FC476A9F983F74DULL); // i= 7
dc64(0x3FEC3F8F01C3F8F0ULL);
dc64(0x3FC70742D4EF027FULL); // i= 8
dc64(0x3FEBDD2B899406F7ULL);
dc64(0x3FC98EDD077E70DFULL); // i= 9
dc64(0x3FEB7D6C3DDA338BULL);
dc64(0x3FCC0DB6CDD94DEEULL); // i=10
dc64(0x3FEB2036406C80D9ULL);
dc64(0x3FCE840BE74E6A4DULL); // i=11
dc64(0x3FEAC5701AC5701BULL);
dc64(0x3FD0790ADBB03009ULL); // i=12
dc64(0x3FEA6D01A6D01A6DULL);
dc64(0x3FD1AC05B291F070ULL); // i=13
dc64(0x3FEA16D3F97A4B02ULL);
dc64(0x3FD2DB10FC4D9AAFULL); // i=14
dc64(0x3FE9C2D14EE4A102ULL);
dc64(0x3FD406463B1B0449ULL); // i=15
dc64(0x3FE970E4F80CB872ULL);
dc64(0x3FD52DBDFC4C96B3ULL); // i=16
dc64(0x3FE920FB49D0E229ULL);
dc64(0x3FD6518FE4677BA7ULL); // i=17
dc64(0x3FE8D3018D3018D3ULL);
dc64(0x3FD771D2BA7EFB3CULL); // i=18
dc64(0x3FE886E5F0ABB04AULL);
dc64(0x3FD88E9C72E0B226ULL); // i=19
dc64(0x3FE83C977AB2BEDDULL);
dc64(0x3FD9A802391E232FULL); // i=20
dc64(0x3FE7F405FD017F40ULL);
dc64(0x3FDABE18797F1F49ULL); // i=21
dc64(0x3FE7AD2208E0ECC3ULL);
dc64(0x3FDBD0F2E9E79031ULL); // i=22
dc64(0x3FE767DCE434A9B1ULL);
dc64(0x3FDCE0A4923A587DULL); // i=23
dc64(0x3FE724287F46DEBCULL);
dc64(0x3FDDED3FD442364CULL); // i=24
dc64(0x3FE6E1F76B4337C7ULL);
dc64(0x3FDEF6D67328E220ULL); // i=25
dc64(0x3FE6A13CD1537290ULL);
dc64(0x3FDFFD799A83FF9BULL); // i=26
dc64(0x3FE661EC6A5122F9ULL);
dc64(0x3FE0809CF27F703DULL); // i=27
dc64(0x3FE623FA77016240ULL);
dc64(0x3FE10113B153C8EAULL); // i=28
dc64(0x3FE5E75BB8D015E7ULL);
dc64(0x3FE18028CF72976AULL); // i=29
dc64(0x3FE5AC056B015AC0ULL);
dc64(0x3FE1FDE3D30E8126ULL); // i=30
dc64(0x3FE571ED3C506B3AULL);
dc64(0x3FE27A4C0585CBF8ULL); // i=31
dc64(0x3FE5390948F40FEBULL);
dc64(0x3FE2F56875EB3F26ULL); // i=32
dc64(0x3FE5015015015015ULL);
dc64(0x3FE36F3FFB6D9162ULL); // i=33
dc64(0x3FE4CAB88725AF6EULL);
dc64(0x3FE3E7D9379F7016ULL); // i=34
dc64(0x3FE49539E3B2D067ULL);
dc64(0x3FE45F3A98A20739ULL); // i=35
dc64(0x3FE460CBC7F5CF9AULL);
dc64(0x3FE4D56A5B33CEC4ULL); // i=36
dc64(0x3FE42D6625D51F87ULL);
dc64(0x3FE54A6E8CA5438EULL); // i=37
dc64(0x3FE3FB013FB013FBULL);
dc64(0x3FE5BE4D0CB51435ULL); // i=38
dc64(0x3FE3C995A47BABE7ULL);
dc64(0x3FE6310B8F553048ULL); // i=39
dc64(0x3FE3991C2C187F63ULL);
dc64(0x3FE6A2AF9E5A0F0AULL); // i=40
dc64(0x3FE3698DF3DE0748ULL);
dc64(0x3FE7133E9B156C7CULL); // i=41
dc64(0x3FE33AE45B57BCB2ULL);
dc64(0x3FE782BDBFDDA657ULL); // i=42
dc64(0x3FE30D190130D190ULL);
dc64(0x3FE7F1322182CF16ULL); // i=43
dc64(0x3FE2E025C04B8097ULL);
dc64(0x3FE85EA0B0B27B26ULL); // i=44
dc64(0x3FE2B404AD012B40ULL);
dc64(0x3FE8CB0E3B4B3BBEULL); // i=45
dc64(0x3FE288B01288B013ULL);
dc64(0x3FE9367F6DA0AB2FULL); // i=46
dc64(0x3FE25E22708092F1ULL);
dc64(0x3FE9A0F8D3B0E050ULL); // i=47
dc64(0x3FE23456789ABCDFULL);
dc64(0x3FEA0A7EDA4C112DULL); // i=48
dc64(0x3FE20B470C67C0D9ULL);
dc64(0x3FEA7315D02F20C8ULL); // i=49
dc64(0x3FE1E2EF3B3FB874ULL);
dc64(0x3FEADAC1E711C833ULL); // i=50
dc64(0x3FE1BB4A4046ED29ULL);
dc64(0x3FEB418734A9008CULL); // i=51
dc64(0x3FE19453808CA29CULL);
dc64(0x3FEBA769B39E4964ULL); // i=52
dc64(0x3FE16E0689427379ULL);
dc64(0x3FEC0C6D447C5DD3ULL); // i=53
dc64(0x3FE1485F0E0ACD3BULL);
dc64(0x3FEC7095AE91E1C7ULL); // i=54
dc64(0x3FE12358E75D3033ULL);
dc64(0x3FECD3E6A0CA8907ULL); // i=55
dc64(0x3FE0FEF010FEF011ULL);
dc64(0x3FED3663B27F31D5ULL); // i=56
dc64(0x3FE0DB20A88F4696ULL);
dc64(0x3FED9810643D6615ULL); // i=57
dc64(0x3FE0B7E6EC259DC8ULL);
dc64(0x3FEDF8F02086AF2CULL); // i=58
dc64(0x3FE0953F39010954ULL);
dc64(0x3FEE59063C8822CEULL); // i=59
dc64(0x3FE073260A47F7C6ULL);
dc64(0x3FEEB855F8CA88FBULL); // i=60
dc64(0x3FE05197F7D73404ULL);
dc64(0x3FEF16E281DB7630ULL); // i=61
dc64(0x3FE03091B51F5E1AULL);
dc64(0x3FEF74AEF0EFAFAEULL); // i=62
dc64(0x3FE0101010101010ULL);
dc64(0x3FEFD1BE4C7F2AF9ULL); // i=63
}
void Dispatcher::EmitF64FPREM() {
// JIT-inlined double-precision FPREM (C-library style truncated remainder).
// Input: VTMP1 = dividend (src1), VTMP2 = divisor (src2). Output: VTMP1.
F64FPREMHandlerAddress = GetCursorAddress<uint64_t>();
constexpr auto Scratch = ARMEmitter::VReg::v2;
ARMEmitter::ForwardLabel ReturnX;
ARMEmitter::ForwardLabel MaybeExact;
ARMEmitter::ForwardLabel Fallback;
ARMEmitter::ForwardLabel NonZeroResult;
// Save q2.
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
// save nzcv
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
fmov(ARMEmitter::Size::i64Bit, TMP3, VTMP1.D());
fmov(ARMEmitter::Size::i64Bit, TMP4, VTMP2.D());
fdiv(Scratch.D(), VTMP1.D(), VTMP2.D());
frintz(VTMP1.D(), Scratch.D());
fmov(ARMEmitter::Size::i64Bit, TMP1, VTMP1.D());
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP1, 1);
(void)cbz(ARMEmitter::Size::i64Bit, TMP2, &ReturnX);
fmov(ARMEmitter::Size::i64Bit, TMP2, Scratch.D());
ubfx(ARMEmitter::Size::i64Bit, TMP2, TMP2, 52, 11);
cmp(ARMEmitter::Size::i64Bit, TMP2, 0x7FF);
(void)b(ARMEmitter::Condition::CC_EQ, &Fallback);
cmp(ARMEmitter::Size::i64Bit, TMP2, 1076);
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
fsub(VTMP2.D(), Scratch.D(), VTMP1.D());
fabs(VTMP2.D(), VTMP2.D());
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 53);
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 52);
fmov(ARMEmitter::Size::i64Bit, Scratch.D(), TMP2);
fcmp(VTMP2.D(), Scratch.D());
// err <= 0.5 ULP could mean fdiv rounded across an integer boundary; defer to MaybeExact
// which distinguishes the (safe) exact-quotient case from the (unsafe) rounded case.
(void)b(ARMEmitter::Condition::CC_LS, &MaybeExact);
fmov(ARMEmitter::ScalarRegSize::i64Bit, Scratch, 1.0f);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP2);
fsub(Scratch.D(), Scratch.D(), VTMP1.D());
fcmp(VTMP2.D(), Scratch.D());
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
fmov(ARMEmitter::Size::i64Bit, Scratch.D(), TMP4);
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
fmsub(VTMP2.D(), VTMP1.D(), Scratch.D(), VTMP2.D());
fmov(ARMEmitter::Size::i64Bit, TMP2, VTMP2.D());
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NonZeroResult);
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP3, 63);
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 63);
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP2);
(void)Bind(&NonZeroResult);
fmov(VTMP1.D(), VTMP2.D());
// restore nzcv
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
ret();
(void)Bind(&ReturnX);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
// restore nzcv
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
ret();
// err <= 0.5 ULP. Two cases:
// err == 0: x/y rounded to an exact integer. Either truly exact (q is right) or fdiv
// rounded a near-integer onto an integer (q is off by one). Verify with
// fmsub(q, y, x): if exactly zero, x == q*y exactly => result is sign(x)*0.
// err > 0: q_fp is genuinely between integers but within rounding slack; fall back.
(void)Bind(&MaybeExact);
fmov(ARMEmitter::Size::i64Bit, TMP2, VTMP2.D());
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &Fallback);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
fmov(ARMEmitter::Size::i64Bit, Scratch.D(), TMP4);
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
fmsub(VTMP2.D(), VTMP1.D(), Scratch.D(), VTMP2.D());
fmov(ARMEmitter::Size::i64Bit, TMP2, VTMP2.D());
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &Fallback);
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP3, 63);
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 63);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP2);
// restore nzcv
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
ret();
(void)Bind(&Fallback);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP4);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FPREM].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FPREM].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
ret();
}
void Dispatcher::EmitF64FPREM1() {
// JIT-inlined double-precision FPREM1 (IEEE round-to-nearest remainder).
// Input: VTMP1 = dividend (src1), VTMP2 = divisor (src2). Output: VTMP1.
F64FPREM1HandlerAddress = GetCursorAddress<uint64_t>();
constexpr auto Scratch = ARMEmitter::VReg::v2;
ARMEmitter::ForwardLabel ReturnX;
ARMEmitter::ForwardLabel Fallback;
ARMEmitter::ForwardLabel NonZeroResult;
// Save q2.
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
// save nzcv
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
fmov(ARMEmitter::Size::i64Bit, TMP3, VTMP1.D());
fmov(ARMEmitter::Size::i64Bit, TMP4, VTMP2.D());
fdiv(Scratch.D(), VTMP1.D(), VTMP2.D());
frintn(VTMP1.D(), Scratch.D());
fmov(ARMEmitter::Size::i64Bit, TMP1, VTMP1.D());
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP1, 1);
(void)cbz(ARMEmitter::Size::i64Bit, TMP2, &ReturnX);
fmov(ARMEmitter::Size::i64Bit, TMP2, Scratch.D());
ubfx(ARMEmitter::Size::i64Bit, TMP2, TMP2, 52, 11);
cmp(ARMEmitter::Size::i64Bit, TMP2, 0x7FF);
(void)b(ARMEmitter::Condition::CC_EQ, &Fallback);
cmp(ARMEmitter::Size::i64Bit, TMP2, 1076);
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
fsub(VTMP2.D(), Scratch.D(), VTMP1.D());
fabs(VTMP2.D(), VTMP2.D());
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 53);
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 52);
fmov(ARMEmitter::ScalarRegSize::i64Bit, Scratch, 0.5f);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP2);
fsub(Scratch.D(), Scratch.D(), VTMP1.D());
fcmp(VTMP2.D(), Scratch.D());
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
fmov(ARMEmitter::Size::i64Bit, Scratch.D(), TMP4);
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
fmsub(VTMP2.D(), VTMP1.D(), Scratch.D(), VTMP2.D());
fmov(ARMEmitter::Size::i64Bit, TMP2, VTMP2.D());
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NonZeroResult);
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP3, 63);
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 63);
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP2);
(void)Bind(&NonZeroResult);
fmov(VTMP1.D(), VTMP2.D());
// restore nzcv
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
ret();
(void)Bind(&ReturnX);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
// restore nzcv
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
ret();
(void)Bind(&Fallback);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP4);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FPREM1].ABIHandler));
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FPREM1].Func));
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
ret();
(void)Bind(&BiasLabel);
dc64(0x3FF0'0000'0000'0000ULL);
(void)Bind(&Sqrt2Label);
dc64(0x3FF6'A09E'667F'3BCDULL);
(void)Bind(&Log2eLabel);
dc64(0x3FF7'1547'652B'82FEULL);
(void)Bind(&P8Label);
dc64(0x3FAA'F286'BCA1'AF28ULL);
(void)Bind(&P7Label);
dc64(0x3FAE'1E1E'1E1E'1E1EULL);
(void)Bind(&P6Label);
dc64(0x3FB1'1111'1111'1111ULL);
(void)Bind(&P5Label);
dc64(0x3FB3'B13B'13B1'3B14ULL);
(void)Bind(&P4Label);
dc64(0x3FB7'45D1'745D'1746ULL);
(void)Bind(&P3Label);
dc64(0x3FBC'71C7'1C71'C71CULL);
(void)Bind(&P2Label);
dc64(0x3FC2'4924'9249'2492ULL);
(void)Bind(&P1Label);
dc64(0x3FC9'9999'9999'999AULL);
(void)Bind(&P0Label);
dc64(0x3FD5'5555'5555'5555ULL);
}
uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
@@ -2685,8 +2176,6 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
Ptrs.ExitFunctionLinker = ExitFunctionLinkerAddress;
Ptrs.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
Ptrs.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
Ptrs.ThreadDispatchSyscallHandler = ThreadDispatchSyscallHandler;
Ptrs.ThreadDispatchRemoveCodeEntry = ThreadDispatchRemoveCodeEntry;
Ptrs.GuestSignal_SIGILL = GuestSignal_SIGILL;
Ptrs.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
Ptrs.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
@@ -2701,9 +2190,6 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
Ptrs.F64ScaleHandler = F64ScaleHandlerAddress;
Ptrs.F64AtanHandler = F64AtanHandlerAddress;
Ptrs.F64FYL2XHandler = F64FYL2XHandlerAddress;
Ptrs.F64FYL2XP1Handler = F64FYL2XP1HandlerAddress;
Ptrs.F64FPREMHandler = F64FPREMHandlerAddress;
Ptrs.F64FPREM1Handler = F64FPREM1HandlerAddress;
// Fill in the fallback handlers
InterpreterOps::FillFallbackIndexPointers(Ptrs.FallbackHandlerPointers, &ABIPointers[0]);
@@ -81,8 +81,6 @@ private:
uint64_t AbsoluteLoopTopAddressEnterECFillSRA {};
uint64_t ThreadPauseHandlerAddress {};
uint64_t ThreadPauseHandlerAddressSpillSRA {};
uint64_t ThreadDispatchSyscallHandler {};
uint64_t ThreadDispatchRemoveCodeEntry {};
uint64_t ExitFunctionLinkerAddress {};
uint64_t SignalHandlerReturnAddress {};
uint64_t SignalHandlerReturnAddressRT {};
@@ -109,9 +107,6 @@ private:
uint64_t F64ScaleHandlerAddress {};
uint64_t F64AtanHandlerAddress {};
uint64_t F64FYL2XHandlerAddress {};
uint64_t F64FYL2XP1HandlerAddress {};
uint64_t F64FPREMHandlerAddress {};
uint64_t F64FPREM1HandlerAddress {};
void EmitDispatcher();
uint64_t GenerateABICall(FallbackABI ABI);
@@ -123,25 +118,13 @@ private:
void EmitF32ToExtF80();
void EmitF64ToExtF80();
// Shared label set for the LUT-based F64 log2 path used by both FYL2X and
// FYL2XP1. The pool is emitted once via EmitF64Log2Constants.
struct F64Log2Constants {
ARMEmitter::ForwardLabel One;
ARMEmitter::ForwardLabel A0, A1, A2, A3, A4, A5, A6, A7;
ARMEmitter::ForwardLabel Table;
};
void EmitF64Sin();
void EmitF64Cos();
void EmitF64Tan();
void EmitF64F2XM1();
void EmitF64Scale();
void EmitF64Atan();
void EmitF64FYL2X(F64Log2Constants& C);
void EmitF64FYL2XP1(F64Log2Constants& C);
void EmitF64Log2Constants(F64Log2Constants& C);
void EmitF64FPREM();
void EmitF64FPREM1();
void EmitF64FYL2X();
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
};
+151 -386
View File
@@ -8,7 +8,6 @@ $end_info$
#include "Interface/Context/Context.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include "Interface/Core/LookupCache.h"
@@ -70,6 +69,7 @@ static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
: Thread {Thread}
, CTX {static_cast<FEXCore::Context::ContextImpl*>(Thread->CTX)}
, OSABI {CTX->SyscallHandler ? CTX->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN}
, PoolObject {CTX->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
@@ -89,11 +89,6 @@ Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
}
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
// Check for wraparound
if (Address + Size < Address) {
return false;
}
while (Address < ExecutableRangeBase || Address + Size > ExecutableRangeEnd) {
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, Address);
ExecutableRangeBase = RangeInfo.Base;
@@ -115,8 +110,9 @@ bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
}
uint8_t Decoder::ReadByte() {
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
std::optional<uint8_t> Byte = PeekByte(0);
if (!Byte || InstructionSize == MAX_INST_SIZE) {
if (!Byte) {
HitNonExecutableRange = true;
// Pretend we read 0, the main decode loop will see HitNonExecutableRange and rollback the instruction.
return 0;
@@ -128,9 +124,9 @@ uint8_t Decoder::ReadByte() {
}
std::optional<uint8_t> Decoder::PeekByte(uint8_t Offset) {
uint64_t ByteAddress = reinterpret_cast<uint64_t>(InstStream.InstStream + InstructionSize + Offset);
uint64_t ByteAddress = reinterpret_cast<uint64_t>(InstStream + InstructionSize + Offset);
if (CheckRangeExecutable(ByteAddress, 1)) {
return InstStream.AdjustedInstStream[InstructionSize + Offset];
return InstStream[InstructionSize + Offset];
} else {
return std::nullopt;
}
@@ -140,11 +136,9 @@ std::pair<uint64_t, bool> Decoder::ReadData(uint8_t Size) {
LOGMAN_THROW_A_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
uint64_t Res = 0;
uint64_t Address = reinterpret_cast<uint64_t>(InstStream.InstStream + InstructionSize);
LastFieldReadOffset = (uint8_t)InstructionSize;
LastFieldReadSize = Size;
uint64_t Address = reinterpret_cast<uint64_t>(InstStream + InstructionSize);
if (CheckRangeExecutable(Address, Size)) {
std::memcpy(&Res, &InstStream.AdjustedInstStream[InstructionSize], Size);
std::memcpy(&Res, &InstStream[InstructionSize], Size);
} else {
HitNonExecutableRange = true;
// See PeekByte, this specific case may cause some executable memory to read as 0 but it doesn't matter as the entire instruction will be rolled back anyway.
@@ -217,7 +211,6 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModR
Operand->Type = DecodedOperand::OpType::SIB;
Operand->Data.SIB.Scale = 1;
Operand->Data.SIB.Offset = Literal;
Operand->Data.SIB.PatchableDisp = false;
// Only called when ModRM.mod != 0b11
struct Encodings {
@@ -293,7 +286,6 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
// SIB
Operand->Type = DecodedOperand::OpType::SIB;
Operand->Data.SIB.Scale = 1 << SIB.scale;
Operand->Data.SIB.PatchableDisp = false;
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
{
@@ -332,7 +324,6 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
auto [Literal, IsRelocation] = ReadData(4);
Operand->Type = IsRelocation ? DecodedOperand::OpType::RIPRelativeRelocation : DecodedOperand::OpType::RIPRelative;
Operand->Data.RIPLiteral.Value = Literal;
Operand->Data.RIPLiteral.PatchableDisp = false;
} else {
// Register-direct addressing
Operand->Type = DecodedOperand::OpType::GPRDirect;
@@ -348,11 +339,10 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
Operand->Type = IsRelocation ? DecodedOperand::OpType::GPRIndirectRelocation : DecodedOperand::OpType::GPRIndirect;
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
Operand->Data.GPRIndirect.Displacement = Literal;
Operand->Data.GPRIndirect.PatchableDisp = false;
}
}
Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options) {
bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options) {
if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) [[unlikely]] {
// Dispatcher Op.
// TODO: Move this in to `NormalOpHeader`, Dispatch tables have a bug currently where some subtables don't inherit flags correctly.
@@ -364,16 +354,11 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
DecodeInst->TableInfo = Info;
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
return DecodedBlockStatus::INVALID_INST;
}
if (!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SUPPORTS_LOCK) && (DecodeInst->Flags & DecodeFlags::FLAG_LOCK)) {
// Instruction has lock prefix but doesn't support lock.
return DecodedBlockStatus::UNIMPLEMENTED_INST;
return false;
}
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P), "Group Ops "
@@ -405,15 +390,15 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
const bool Has16BitAddressing = !BlockInfo.Is64BitMode && DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
if (Options.w && (Info->Flags & InstFlags::FLAGS_REX_W_0)) {
return DecodedBlockStatus::INVALID_INST;
return false;
} else if (!Options.w && (Info->Flags & InstFlags::FLAGS_REX_W_1)) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
if (Options.L && (Info->Flags & InstFlags::FLAGS_VEX_L_0)) {
return DecodedBlockStatus::INVALID_INST;
return false;
} else if (!Options.L && (Info->Flags & InstFlags::FLAGS_VEX_L_1)) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
const bool UseVEXL = Options.L && !(Info->Flags & InstFlags::FLAGS_VEX_L_IGNORE);
@@ -501,7 +486,15 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
auto* CurrentDest = &DecodeInst->Dest;
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RDX)) {
// Some instructions hardcode their destination as RAX
CurrentDest->Type = DecodedOperand::OpType::GPR;
CurrentDest->Data.GPR.HighBits = false;
CurrentDest->Data.GPR.GPR =
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
CurrentDest = &DecodeInst->Src[0];
} else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
@@ -514,7 +507,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
if (CurrentDest->Data.GPR.GPR == FEXCore::X86State::REG_INVALID) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
}
@@ -583,7 +576,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
const auto VEXOperand = Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_SRC_MASK;
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_NO_OPERAND && Options.vvvv) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) {
@@ -601,11 +594,11 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM) {
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SF_MOD_DST) {
if (!ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest)) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
} else {
if (!ModRMOperand(DecodeInst->Dest, DecodeInst->Src[CurrentSrc], HasXMMDst, HasXMMSrc, HasMMDst, HasMMSrc, Is8BitDest, Is8BitSrc)) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
}
++CurrentSrc;
@@ -618,13 +611,28 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
++CurrentSrc;
}
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RAX)) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RAX;
++CurrentSrc;
} else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RCX)) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RCX;
++CurrentSrc;
}
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) {
CurrentDest->Type = DecodedOperand::OpType::GPR;
CurrentDest->Data.GPR.HighBits = false;
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
}
if (Bytes <= 8 && Bytes > 0) {
if (Bytes != 0) {
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
auto [Literal, IsRelocation] = ReadData(Bytes);
if (IsRelocation) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::LiteralRelocation;
@@ -650,34 +658,24 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
}
Bytes = 0;
} else {
// All real x86 instructions have byte sizes that are 8-bytes or less.
// Thunk instruction has an additional 32-byte SHA256 payload that needs to be accounted for.
InstructionSize += Bytes;
Bytes = 0;
}
if ((DecodeInst->Flags & DecodeFlags::FLAG_LOCK) && DecodeInst->Dest.IsGPR()) {
// Instruction has lock prefix, but the destination isn't memory, this is invalid.
return DecodedBlockStatus::UNIMPLEMENTED_INST;
}
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
DecodeInst->InstSize = InstructionSize;
return DecodedBlockStatus::SUCCESS;
return true;
}
Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op) {
bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op) {
DecodeInst->OPRaw = DecodeInst->OP = Op;
DecodeInst->TableInfo = Info;
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
@@ -734,7 +732,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
};
uint8_t Field = RegToField[ModRM.reg];
if (Field == 255) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
LocalOp = (Field << 3) | ModRM.rm;
@@ -753,7 +751,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
} else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
if (!VEXTable) {
// AVX not enabled.
return DecodedBlockStatus::INVALID_INST;
return false;
}
uint16_t map_select = 1;
@@ -763,7 +761,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
if ((Byte1 & 0b10000000) == 0) {
if (!BlockInfo.Is64BitMode) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
@@ -774,7 +772,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
const uint8_t vvvv = ((Byte1 & 0b01111000) >> 3);
if (!BlockInfo.Is64BitMode && vvvv <= 0b0111) {
// Invalid on 32-bit, can't use the high registers.
return DecodedBlockStatus::INVALID_INST;
return false;
}
options.vvvv = 15 - vvvv;
options.L = (Byte1 & 0b100) != 0;
@@ -785,14 +783,14 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
const uint8_t vvvv = ((Byte2 & 0b01111000) >> 3);
if (!BlockInfo.Is64BitMode && vvvv <= 0b0111) {
// Invalid on 32-bit, can't use the high registers.
return DecodedBlockStatus::INVALID_INST;
return false;
}
options.vvvv = 15 - vvvv;
options.w = (Byte2 & 0b10000000) != 0;
options.L = (Byte2 & 0b100) != 0;
if ((Byte1 & 0b01000000) == 0) {
if (!BlockInfo.Is64BitMode) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
}
@@ -803,7 +801,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
DecodeInst->Flags |= DecodeFlags::FLAG_OPTION_AVX_W;
}
if (!(map_select >= 1 && map_select <= 3)) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
}
@@ -833,14 +831,14 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
} else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
FEXCORE_TELEMETRY_SET(TYPE_USES_EVEX_OPS, 1);
// EVEX unsupported
return DecodedBlockStatus::INVALID_INST;
return false;
}
LOGMAN_MSG_A_FMT("Invalid instruction decoding type");
FEX_UNREACHABLE;
}
Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
InstructionSize = 0;
LastEscapePrefix = 0;
Instruction.fill(0);
@@ -851,7 +849,7 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
for (;;) {
if (InstructionSize >= MAX_INST_SIZE) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
uint8_t Op = ReadByte();
switch (Op) {
@@ -1037,10 +1035,10 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
}
if (DecodeInst->Dest.IsGPR()) {
return DecodedBlockStatus::INVALID_INST;
return false;
}
return DecodedBlockStatus::SUCCESS;
return true;
}
void Decoder::DecodeREXIfValid(int8_t ExpectedOffset) {
@@ -1078,26 +1076,18 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
// Will be set if DecodeInstructionImpl tries to read non-executable memory
HitNonExecutableRange = false;
HitBadRelocation = false;
auto ErrorDuringDecoding = DecodeInstructionImpl(PC);
bool ErrorDuringDecoding = !DecodeInstructionImpl(PC);
if (ErrorDuringDecoding != DecodedBlockStatus::SUCCESS || HitNonExecutableRange || HitBadRelocation) [[unlikely]] {
if (ErrorDuringDecoding || HitNonExecutableRange || HitBadRelocation) [[unlikely]] {
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
// Error while decoding instruction. We don't know the table or instruction size
const auto InstSize = DecodeInst->InstSize;
DecodeInst->TableInfo = nullptr;
auto Result = ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST :
DecodeInst->InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST :
HitNonExecutableRange ? DecodedBlockStatus::NOEXEC_INST :
DecodedBlockStatus::BAD_RELOCATION;
DecodeInst->InstSize = 0;
// A decode error can be caused by substituting zero for an inaccessible
// instruction byte, so the instruction fetch fault takes priority.
if (HitNonExecutableRange) {
return InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST : DecodedBlockStatus::NOEXEC_INST;
}
if (HitBadRelocation) {
return DecodedBlockStatus::BAD_RELOCATION;
}
return ErrorDuringDecoding;
return Result;
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
return DecodedBlockStatus::INVALID_INST;
@@ -1321,13 +1311,6 @@ void Decoder::AddBranchTarget(uint64_t Target) {
.BlockStatus = BlockIt->BlockStatus,
};
if (BlockIt->DataMasks.size()) {
auto MaskIt = std::lower_bound(BlockIt->DataMasks.begin(), BlockIt->DataMasks.end(), SplitAddr,
[](const DataMask& Mask, uint64_t Addr) { return Mask.FieldAddress < Addr; });
SplitBlock.DataMasks.assign(MaskIt, BlockIt->DataMasks.end());
BlockIt->DataMasks.erase(MaskIt, BlockIt->DataMasks.end());
}
BlockIt->Size = SplitOffset;
BlockIt->NumInstructions = SplitIdx;
@@ -1348,242 +1331,125 @@ void Decoder::AddBranchTarget(uint64_t Target) {
}
}
const Decoder::DecodeStream Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
if (BlockInfo.Is64BitMode && CTX->HostFeatures.HostType == FEXCore::HostFeatures::HostTypeEnum::Linux && RIP >= VSyscall_Base &&
RIP < VSyscall_End) {
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64 && RIP >= VSyscall_Base && RIP < VSyscall_End) {
// VSyscall
// This doesn't exist on AArch64 and on x86_64 hosts this is emulated with faults to a region mapped with --xp permissions
// Offset 0: vgettimeofday
// Offset 0x400: vtime
// Offset 0x800: vgetcpu
uint64_t Offset = RIP - VSyscall_Base;
return DecodeStream {
.InstStream = _InstStream - EntryPoint + RIP,
.AdjustedInstStream = VSyscallData + Offset,
};
return VSyscallData + Offset;
}
return DecodeStream {
.InstStream = _InstStream - EntryPoint + RIP,
.AdjustedInstStream = _InstStream - EntryPoint + RIP,
};
return _InstStream - EntryPoint + RIP;
}
bool Decoder::CheckIfCacheable(FEXCore::Core::InternalThreadState& Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst) {
SetupDecodeInstructionsAtEntry(&Thread, PC, MaxInst);
DecodeLoop(InstStream);
DecodeInstructionsAtEntry(&Thread, InstStream, PC, MaxInst);
bool Uncacheable = HitBadRelocation;
DelayedDisownBuffer();
return !Uncacheable;
}
void Decoder::DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block) {
FEXCore::X86Tables::DecodedOperand* LiteralToPatch = nullptr;
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
BlockInfo.TotalInstructionCount = 0;
BlockInfo.Blocks.clear();
VisitedBlocks.clear();
// Reset internal state management
DecodedSize = 0;
MaxCondBranchForward = 0;
MaxCondBranchBackwards = ~0ULL;
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
// cmp *, imm8 - seen varying in mono jitted code
{
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
if ((DecodeInst->OPRaw == 0x80 || DecodeInst->OPRaw == 0x83) && ModRM.reg == 7 && LastFieldReadSize == 1) {
for (auto& Src : DecodeInst->Src) {
if (Src.IsLiteral()) {
LiteralToPatch = &Src;
break;
}
}
if (LiteralToPatch && LiteralToPatch->Literal() != 0) {
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, DataMaskType::MOV, LastFieldReadSize});
// Decode operating mode from thread's CS segment.
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
BlockInfo.Is64BitMode = CSSegment->L == 1;
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
LiteralToPatch->Data.LiteralPatchable.Width = LastFieldReadSize;
}
EntryPoint = PC;
BlockInfo.EntryPoints = {PC};
InstStream = _InstStream;
uint64_t TotalInstructions {};
SectionMinAddress = 0;
SectionMaxAddress = ~0ULL;
Relocations = nullptr;
if (CTX->GetCodeCache().IsGeneratingCache || EnableCodeCacheValidation) {
// If generating cache, attempt to load section bounds and relocations
if (auto SectionInfo = CTX->SyscallHandler->LookupExecutableFileSection(Thread, EntryPoint)) {
SectionMinAddress = SectionInfo->FileStartVA;
SectionMaxAddress = SectionInfo->EndVA;
Relocations = &SectionInfo->FileInfo.Relocations;
}
}
if (LastFieldReadSize < 4) {
return;
DecodedMinAddress = EntryPoint;
DecodedMaxAddress = EntryPoint;
// Entry is a jump target
BlocksToDecode = {PC};
uint64_t CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
BlockInfo.CodePages = {CurrentCodePage};
if (MaxInst == 0) {
MaxInst = CTX->Config.MaxInstPerBlock;
}
DataMaskType Type = DataMaskType::NONE;
bool EntryBlock {true};
bool FinalInstruction {false};
// imm32 or imm64 at the end type instructions
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_LITERAL_PATCHABLE) {
for (auto& Src : DecodeInst->Src) {
if (Src.IsLiteral()) {
LiteralToPatch = &Src;
break;
}
while (!FinalInstruction && !BlocksToDecode.empty()) {
auto BlockDecodeIt = BlocksToDecode.begin();
uint64_t RIPToDecode = *BlockDecodeIt;
BlocksToDecode.erase(BlockDecodeIt);
VisitedBlocks.emplace(RIPToDecode);
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
NextBlockStartAddress = ~0ULL;
if (!BlocksToDecode.empty()) {
// We just erased the lowest, the front is then the second lowest
NextBlockStartAddress = *BlocksToDecode.begin();
}
if (DecodeInst->Dest.IsLiteral()) {
LiteralToPatch = &DecodeInst->Dest;
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
NextBlockStartAddress = BlockSuccIt->Entry;
}
LOGMAN_THROW_A_FMT(NextBlockStartAddress > RIPToDecode, "unexpected");
// heuristic: if it's a small value, assume it's more likely to be part of the code around it
bool IsMOV = (DecodeInst->OPRaw >= 0xB8 && DecodeInst->OPRaw <= 0xBF) || DecodeInst->OPRaw == 0xC7;
if (LiteralToPatch && IsMOV && LiteralToPatch->Literal() < 0x10000ULL) {
LiteralToPatch = nullptr;
}
if (LiteralToPatch && LiteralToPatch->Literal() != 0) {
Type = DataMaskType::MOV;
}
}
// Insert the block now so it can be looked up and split if necessary on a backward edge
auto BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
// anything that has a patchable disp32
bool TryDisp = DecodeInst->Flags & X86Tables::DecodeFlags::FLAG_DECODED_MODRM;
if (TryDisp) {
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
{
// todo we could handle both imm + disp with more load tracking
bool FoundLiteral = false;
FEXCore::X86Tables::DecodedOperand* OpToPatch = nullptr;
for (auto& Src : DecodeInst->Src) {
if (Src.IsLiteral()) {
FoundLiteral = true;
break;
}
if (Src.IsRIPRelative() || Src.IsGPRIndirect() || Src.IsSIB()) {
OpToPatch = &Src;
}
}
if (DecodeInst->Dest.IsLiteral()) {
FoundLiteral = true;
}
if (DecodeInst->Dest.IsRIPRelative() || DecodeInst->Dest.IsGPRIndirect() || DecodeInst->Dest.IsSIB()) {
OpToPatch = &DecodeInst->Dest;
}
if (!FoundLiteral && OpToPatch) {
if (DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch == &IR::OpDispatchBuilder::NOPOp) {
// if it's a nop disp, only mask data out, no other action required
Type = DataMaskType::NOP;
} else if (OpToPatch->IsRIPRelative()) {
Type = DataMaskType::DISP;
OpToPatch->Data.RIPLiteral.PatchableDisp = true;
OpToPatch->Data.RIPLiteral.DispOffset = LastFieldReadOffset;
} else if (OpToPatch->IsGPRIndirect()) {
// filter out disp8
if (!(ModRM.mod == 1)) {
Type = DataMaskType::DISP;
OpToPatch->Data.GPRIndirect.PatchableDisp = true;
OpToPatch->Data.GPRIndirect.DispOffset = LastFieldReadOffset;
}
} else if (OpToPatch->IsSIB()) {
FEXCore::X86Tables::SIBDecoded SIB;
SIB.Hex = DecodeInst->SIB;
// two disp32 cases here
if (ModRM.mod == 0b10 || (ModRM.mod == 0 && SIB.base == 0b101)) {
Type = DataMaskType::DISP;
OpToPatch->Data.SIB.PatchableDisp = true;
OpToPatch->Data.SIB.DispOffset = LastFieldReadOffset;
}
}
}
}
}
// jmp/call branches that use a literal rip-relative offset
// some of those may be inlined by multiblock and will be cleaned up at decode end
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_SETS_RIP && DecodeInst->Src[0].IsLiteral() && DecodeInst->Src[0].Literal() != 0) {
LiteralToPatch = &DecodeInst->Src[0];
Type = DataMaskType::BRANCH;
}
BlockIt->Entry = RIPToDecode;
BlockIt->Size = 0;
BlockIt->IsEntryPoint = EntryBlock;
// todo add a bunch more
uint64_t PCOffset = 0;
uint64_t BlockStartOffset = DecodedSize;
bool EraseBlock = true; // Unset once the block contains an instruction
if (Type != DataMaskType::NONE) {
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, Type, LastFieldReadSize});
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
BlockIt->NumInstructions = 0;
if (LiteralToPatch) {
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
LiteralToPatch->Data.LiteralPatchable.Width = LastFieldReadSize;
}
}
}
void Decoder::PruneInlinedBranchDataMasks() {
for (auto& Block : BlockInfo.Blocks) {
if (!Block.DataMasks.size()) {
continue;
}
const auto& LastInst = Block.DecodedInstructions[Block.NumInstructions - 1];
const auto& LastMask = Block.DataMasks.back();
if (LastMask.Type != DataMaskType::BRANCH) {
continue;
}
const uint64_t NextInst = LastInst.PC + LastInst.InstSize;
if (LastMask.FieldAddress < LastInst.PC || LastMask.FieldAddress + LastMask.ValueSize > NextInst) {
continue;
}
if (std::ranges::binary_search(BlockInfo.Blocks, NextInst + LastInst.Src[0].Data.LiteralPatchable.Value, std::less {}, &DecodedBlocks::Entry)) {
Block.DataMasks.pop_back();
}
}
}
void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
// counter-intuitively, the masks are also needed for lookup on anon prefix decodes, not just stores
bool WantsDataMasks = CTX->DiskCache.IsReadingDiskCache() || CTX->DiskCache.IsWritingDiskCache();
while (!FinalInstruction && (Paused || !BlocksToDecode.empty())) {
bool Pausing = false;
fextl::vector<DecodedBlocks>::iterator BlockIt;
if (!Paused || BlockResume == -1) {
auto BlockDecodeIt = BlocksToDecode.begin();
uint64_t RIPToDecode = *BlockDecodeIt;
BlocksToDecode.erase(BlockDecodeIt);
VisitedBlocks.emplace(RIPToDecode);
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
NextBlockStartAddress = ~0ULL;
if (!BlocksToDecode.empty()) {
// We just erased the lowest, the front is then the second lowest
NextBlockStartAddress = *BlocksToDecode.begin();
}
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
NextBlockStartAddress = BlockSuccIt->Entry;
}
LOGMAN_THROW_A_FMT(NextBlockStartAddress == ~0ULL || NextBlockStartAddress > RIPToDecode, "unexpected");
// Insert the block now so it can be looked up and split if necessary on a backward edge
BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
BlockIt->Entry = RIPToDecode;
BlockIt->Size = 0;
BlockIt->IsEntryPoint = EntryBlock;
PCOffset = 0;
BlockStartOffset = DecodedSize;
EraseBlock = true; // Unset once the block contains an instruction
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
BlockIt->NumInstructions = 0;
// Do a bit of pointer math to figure out where we are in code
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
} else if (BlockResume != -1) {
BlockIt = BlockInfo.Blocks.begin() + BlockResume;
BlockResume = -1;
}
Paused = false;
// Do a bit of pointer math to figure out where we are in code
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
while (1) {
InstructionSize = 0;
// MAX_INST_SIZE assumes worst case
auto OpAddress = BlockIt->Entry + PCOffset;
auto OpAddress = RIPToDecode + PCOffset;
auto OpMaxAddress = OpAddress + MAX_INST_SIZE;
auto OpMinPage = OpAddress & FEXCore::Utils::FEX_PAGE_MASK;
@@ -1605,7 +1471,6 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
BlockInfo.CodePages.insert(CurrentCodePage);
}
LastFieldReadSize = 0;
BlockIt->BlockStatus = DecodeInstruction(OpAddress);
if (HitBadRelocation) {
BlockInfo.TotalInstructionCount = 0;
@@ -1630,11 +1495,6 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
++BlockIt->NumInstructions;
BlockIt->Size += DecodeInst->InstSize;
// if we weren't provided relocations (guest JIT), try to detect what we can
if (WantsDataMasks && BlockIt->BlockStatus == DecodedBlockStatus::SUCCESS && BlockInfo.Is64BitMode && !Relocations) {
DetectDataMasks(OpAddress, *BlockIt);
}
// Can not continue this block at all on invalid instruction
if (BlockIt->BlockStatus != DecodedBlockStatus::SUCCESS) [[unlikely]] {
if (!EntryBlock && BlockIt->BlockStatus != DecodedBlockStatus::BAD_RELOCATION) {
@@ -1644,31 +1504,18 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
TotalInstructions -= BlockIt->NumInstructions;
DecodedSize = BlockStartOffset;
InstStream -= PCOffset;
if (DecodedMaxAddress == OpEndAddress) {
DecodedMaxAddress -= PCOffset;
}
EraseBlock = true;
} else {
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" :
BlockIt->BlockStatus == DecodedBlockStatus::NOEXEC_INST ? "NoExec" :
BlockIt->BlockStatus == DecodedBlockStatus::BAD_RELOCATION ? "BadRelocation" :
BlockIt->BlockStatus == DecodedBlockStatus::UNIMPLEMENTED_INST ? "Unimplemented" :
"PartialDecode",
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" :
BlockIt->BlockStatus == DecodedBlockStatus::NOEXEC_INST ? "NoExec" :
BlockIt->BlockStatus == DecodedBlockStatus::BAD_RELOCATION ? "BadRelocation" :
"PartialDecode",
OpAddress);
}
break;
}
if (GuestSizePause) {
if (GuestSizePause > DecodeInst->InstSize) {
GuestSizePause -= DecodeInst->InstSize;
} else {
GuestSizePause = 0;
Pausing = true;
}
}
// Check if we need to end the entire multiblock
FinalInstruction = DecodedSize >= MaxInst || DecodedSize >= DefaultDecodedBufferSize || TotalInstructions >= MaxInst;
if (FinalInstruction) {
@@ -1681,9 +1528,7 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
// If the branch target is within our multiblock range then we can keep going on
// We don't want to short circuit this since we want to calculate our ranges still
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
if (CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions)) {
BlockIt->ForceFullSMCDetection = true;
}
BlockIt->ForceFullSMCDetection = CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions);
BranchTargetInMultiblockRange();
}
@@ -1692,17 +1537,6 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
PCOffset += DecodeInst->InstSize;
InstStream += DecodeInst->InstSize;
if (Pausing) {
Pausing = false;
Paused = true;
BlockResume = BlockIt - BlockInfo.Blocks.begin();
break;
}
}
if (Paused) {
break;
}
// NOTE: BlockIt is only valid here in the EraseBlock case
@@ -1714,16 +1548,6 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
CurrentBlockTargets.clear();
EntryBlock = false;
if (Pausing && !BlocksToDecode.empty() && !FinalInstruction) {
Paused = true;
BlockResume = -1;
break;
}
}
if (Paused) {
return;
}
BlockInfo.TotalInstructionCount = TotalInstructions;
@@ -1731,65 +1555,6 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
for (auto& Block : BlockInfo.Blocks) {
Block.IsEntryPoint = BlockInfo.EntryPoints.contains(Block.Entry);
}
// now that multiblock has settled down, remove any branch masks we put down that didn't end the block
if (WantsDataMasks) {
PruneInlinedBranchDataMasks();
}
}
void Decoder::SetupDecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t PC, uint64_t MaxInst) {
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
BlockInfo.TotalInstructionCount = 0;
BlockInfo.Blocks.clear();
VisitedBlocks.clear();
// Reset internal state management
Paused = false;
BlockResume = -1;
DecodedSize = 0;
if (MaxInst == 0) {
MaxInst = CTX->Config.MaxInstPerBlock;
}
this->MaxInst = MaxInst;
MaxCondBranchForward = 0;
MaxCondBranchBackwards = ~0ULL;
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
// Decode operating mode from thread's CS segment.
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
BlockInfo.Is64BitMode = CSSegment->L == 1;
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
EntryPoint = PC;
BlockInfo.EntryPoints = {PC};
TotalInstructions = 0;
SectionMinAddress = 0;
SectionMaxAddress = ~0ULL;
Relocations = nullptr;
if (CTX->GetCodeCache().IsGeneratingCache || EnableCodeCacheValidation) {
// If generating cache, attempt to load section bounds and relocations
if (auto SectionInfo = CTX->SyscallHandler->LookupExecutableFileSection(Thread, EntryPoint)) {
SectionMinAddress = SectionInfo->FileStartVA;
SectionMaxAddress = SectionInfo->EndVA;
Relocations = &SectionInfo->FileInfo.Relocations;
}
}
DecodedMinAddress = EntryPoint;
DecodedMaxAddress = EntryPoint;
// Entry is a jump target
BlocksToDecode = {PC};
CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
BlockInfo.CodePages = {CurrentCodePage};
EntryBlock = true;
FinalInstruction = false;
}
} // namespace FEXCore::Frontend
+10 -59
View File
@@ -19,6 +19,9 @@
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::HLE {
enum class SyscallOSABI;
}
namespace FEXCore::Frontend {
class Decoder final {
@@ -29,15 +32,6 @@ public:
NOEXEC_INST,
PARTIAL_DECODE_INST,
BAD_RELOCATION,
UNIMPLEMENTED_INST,
};
enum class DataMaskType : uint8_t { NONE, MOV, BRANCH, DISP, NOP };
struct DataMask final {
uint64_t FieldAddress;
DataMaskType Type;
uint8_t ValueSize;
};
// New Frontend decoding
@@ -49,7 +43,6 @@ public:
DecodedBlockStatus BlockStatus;
bool IsEntryPoint {};
bool ForceFullSMCDetection {};
fextl::vector<DataMask> DataMasks;
};
struct DecodedBlockInformation final {
@@ -62,9 +55,7 @@ public:
Decoder(FEXCore::Core::InternalThreadState* Thread);
bool CheckIfCacheable(FEXCore::Core::InternalThreadState&, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
void SetupDecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t PC, uint64_t MaxInst);
void DecodeLoop(const uint8_t* InstStream, uint64_t GuestPause = 0);
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
const DecodedBlockInformation* GetDecodedBlockInfo() const {
return &BlockInfo;
@@ -81,10 +72,6 @@ public:
PoolObject.DelayedDisownBuffer();
}
void ValidateDisownedOrFree() const {
PoolObject.ValidateDisownedOrFree();
}
void ResetExecutableRangeCache() {
ExecutableRangeBase = ExecutableRangeEnd = 0;
}
@@ -100,10 +87,11 @@ private:
FEXCore::Core::InternalThreadState* Thread;
FEXCore::Context::ContextImpl* CTX;
const FEXCore::HLE::SyscallOSABI OSABI {};
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
DecodedBlockStatus DecodeInstructionImpl(uint64_t PC);
bool DecodeInstructionImpl(uint64_t PC);
DecodedBlockStatus DecodeInstruction(uint64_t PC);
void BranchTargetInMultiblockRange();
@@ -112,9 +100,6 @@ private:
void AddBranchTarget(uint64_t Target);
void DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block);
void PruneInlinedBranchDataMasks();
bool CheckRangeExecutable(uint64_t Address, uint64_t Size);
uint8_t ReadByte();
@@ -125,8 +110,8 @@ private:
InstructionSize += Size;
}
DecodedBlockStatus NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
DecodedBlockStatus NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
bool NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
void DecodeREXIfValid(int8_t ExpectedOffset = -1);
@@ -134,19 +119,6 @@ private:
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
Utils::PoolBufferWithTimedRetirement<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
size_t DecodedSize {};
uint64_t TotalInstructions {};
uint64_t CurrentCodePage {};
bool EntryBlock {};
bool FinalInstruction {};
uint64_t MaxInst {};
bool Paused {};
int64_t BlockResume = -1;
uint64_t PCOffset {};
uint64_t BlockStartOffset {};
bool EraseBlock {};
uint8_t LastFieldReadOffset;
uint8_t LastFieldReadSize;
uint64_t ExecutableRangeBase {};
uint64_t ExecutableRangeEnd {};
@@ -154,34 +126,13 @@ private:
bool HitNonExecutableRange {};
bool HitBadRelocation {};
struct DecodeStream {
// Original instruction stream RIP location.
const uint8_t* InstStream;
// Adjusted location for FEX actually decodes from.
const uint8_t* AdjustedInstStream;
DecodeStream& operator-=(size_t offset) noexcept {
InstStream -= offset;
AdjustedInstStream -= offset;
return *this;
}
DecodeStream& operator+=(size_t offset) noexcept {
InstStream += offset;
AdjustedInstStream += offset;
return *this;
}
};
DecodeStream InstStream;
const uint8_t* InstStream {};
IR::OpSize GetGPROpSize() const {
return BlockInfo.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
}
static constexpr size_t MAX_INST_SIZE = 15;
uint8_t InstructionSize {};
// Contains the full decoded instruction, unless it is a `Thunk` instruction.
std::array<uint8_t, MAX_INST_SIZE> Instruction;
uint8_t LastEscapePrefix {};
FEXCore::X86Tables::DecodedInst* DecodeInst;
@@ -217,6 +168,6 @@ private:
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_TABLE_SIZE>* VEXTable {};
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_GROUP_TABLE_SIZE>* VEXTableGroup {};
const DecodeStream AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
};
} // namespace FEXCore::Frontend
@@ -302,16 +302,6 @@ struct OpHandlers<IR::OP_F80FYL2X> {
}
};
template<>
struct OpHandlers<IR::OP_F80FYL2XP1> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
const X80SoftFloat One {&State.State, 1.0};
return X80SoftFloat::FYL2X(&State.State, X80SoftFloat::FADD(&State.State, Src1, One), Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80ATAN> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
@@ -427,14 +417,6 @@ struct OpHandlers<IR::OP_F64FYL2X> {
}
};
template<>
struct OpHandlers<IR::OP_F64FYL2XP1> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return src2 * log2(1.0 + src1);
}
};
template<>
struct OpHandlers<IR::OP_F64SCALE> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
@@ -10,6 +10,11 @@
namespace FEXCore::CPU {
template<typename R, typename... Args>
static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_UNKNOWN, HandlerIndex};
}
void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint64_t* ABIHandlers) {
Info[Core::OPINDEX_F80CVTTO_4] = {ABIHandlers[FABI_F80_I16_F32_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4)};
@@ -67,8 +72,6 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80DIV>::handle)};
Info[Core::OPINDEX_F80FYL2X] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2X>::handle)};
Info[Core::OPINDEX_F80FYL2XP1] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2XP1>::handle)};
Info[Core::OPINDEX_F80ATAN] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80ATAN>::handle)};
Info[Core::OPINDEX_F80FPREM1] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
@@ -94,13 +97,10 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle)};
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle)};
Info[Core::OPINDEX_F64FYL2XP1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2XP1>::handle)};
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle)};
// SSE4.2 string instructions
// NOTE: Currently unused. See VectorFallbacks.h.
Info[Core::OPINDEX_VPCMPESTRX] = {ABIHandlers[FABI_I32_I64_I64_V128_V128_I16],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle)};
Info[Core::OPINDEX_VPCMPISTRX] = {ABIHandlers[FABI_I32_V128_V128_I16],
@@ -212,6 +212,12 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
return true; \
}
#define COMMON_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
return true; \
}
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
@@ -248,7 +254,6 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
COMMON_BINARY_X87_OP(MUL)
COMMON_BINARY_X87_OP(DIV)
COMMON_BINARY_X87_OP(FYL2X)
COMMON_BINARY_X87_OP(FYL2XP1)
COMMON_BINARY_X87_OP(ATAN)
COMMON_BINARY_X87_OP(FPREM1)
COMMON_BINARY_X87_OP(FPREM)
@@ -263,7 +268,6 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
// Double Precision Binary
COMMON_BINARY_F64_OP(FYL2X)
COMMON_BINARY_F64_OP(FYL2XP1)
COMMON_BINARY_F64_OP(ATAN)
COMMON_BINARY_F64_OP(FPREM1)
COMMON_BINARY_F64_OP(FPREM)
@@ -8,7 +8,6 @@
#include <cstring>
// NOTE: Currently unused. See VectorFallbacks.h.
namespace FEXCore::CPU {
#ifdef ARCHITECTURE_arm64
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
@@ -13,33 +13,29 @@
namespace FEXCore::CPU {
// PCMPXSTRX control byte fields
enum class AggregationOp {
EqualAny = 0b00,
Ranges = 0b01,
EqualEach = 0b10,
EqualOrdered = 0b11,
};
enum class SourceData {
U8,
U16,
S8,
S16,
};
enum class Polarity {
Positive,
Negative,
PositiveMasked,
NegativeMasked,
};
// NB: The fallback handlers for the *PMCP*STR* instructions
// are no longer used since we now emit inline ASM for them.
// We preserve this as a reference implementation.
template<>
struct OpHandlers<IR::OP_VPCMPESTRX> {
enum class AggregationOp {
EqualAny = 0b00,
Ranges = 0b01,
EqualEach = 0b10,
EqualOrdered = 0b11,
};
enum class SourceData {
U8,
U16,
S8,
S16,
};
enum class Polarity {
Positive,
Negative,
PositiveMasked,
NegativeMasked,
};
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(uint64_t RAX, uint64_t RDX, VectorRegType lhs_v, VectorRegType rhs_v, uint16_t control) {
__uint128_t lhs;
memcpy(&lhs, &lhs_v, sizeof(lhs));
+8 -21
View File
@@ -9,11 +9,13 @@ $end_info$
#include "FEXCore/IR/IR.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/Core/JIT/Relocations.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
#define DEF_BINOP_WITH_CONSTANT(FEXOp, VarOp, ConstOp) \
DEF_OP(FEXOp) { \
auto Op = IROp->C<IR::IROp_##FEXOp>(); \
@@ -65,21 +67,6 @@ DEF_OP(EntrypointOffset) {
InsertGuestRIPMove(GetReg(Node), Constant & Mask);
}
DEF_OP(PatchableGuestData) {
auto Op = IROp->C<IR::IROp_PatchableGuestData>();
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE, GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
}
DEF_OP(PatchableGuestRIP) {
auto Op = IROp->C<IR::IROp_PatchableGuestRIP>();
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE, GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
}
DEF_OP(PatchableGuestCRC) {
auto Op = IROp->C<IR::IROp_PatchableGuestRIP>();
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_CRC_MOVE, GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
}
DEF_OP(InlineConstant) {
// nop
}
@@ -434,8 +421,8 @@ DEF_OP(MulH) {
if (OpSize == IR::OpSize::i32Bit) {
sxtw(TMP1, Src1.W());
sxtw(TMP2, Src2.W());
mul(ARMEmitter::Size::i64Bit, Dst, TMP1, TMP2);
ubfx(ARMEmitter::Size::i64Bit, Dst, Dst, 32, 32);
mul(ARMEmitter::Size::i32Bit, Dst, TMP1, TMP2);
ubfx(ARMEmitter::Size::i32Bit, Dst, Dst, 32, 32);
} else {
smulh(Dst.X(), Src1.X(), Src2.X());
}
@@ -536,7 +523,7 @@ DEF_OP(AndWithFlags) {
}
DEF_OP(AndShift) {
auto Op = IROp->C<IR::IROp_AndShift>();
auto Op = IROp->C<IR::IROp_XorShift>();
and_(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
@@ -734,7 +721,7 @@ DEF_OP(Extr) {
}
DEF_OP(PDep) {
auto Op = IROp->C<IR::IROp_PDep>();
auto Op = IROp->C<IR::IROp_PExt>();
const auto EmitSize = ConvertSize48(IROp);
const auto Dest = GetReg(Node);
@@ -787,7 +774,7 @@ DEF_OP(PDep) {
// Now, they're copied, so we can start setting Dest (even if it overlaps with
// one of them). Handle early exit case
mov(EmitSize, Dest, 0);
(void)cbz(EmitSize, Mask, &Done);
(void)cbz(EmitSize, OrigMask, &Done);
// Setup for first iteration
neg(EmitSize, T0, Mask);
@@ -58,8 +58,7 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit) {
switch (Lit.MoveABI.Header.Type) {
case RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL:
case RelocationTypes::RELOC_GUEST_RIP_LITERAL:
case RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
case RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
Lit.MoveABI.Header.Offset = GetCursorOffset();
break;
}
@@ -103,37 +102,6 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
Relocations.emplace_back(MoveABI);
}
auto Arm64JITCore::InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize) -> NamedSymbolLiteralPair {
return {
.Lit = GuestRIP,
.MoveABI =
{
.GuestPatchableData = {.Header =
{
.Offset = 0, // Set by PlaceNamedSymbolLiteral
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL,
},
.RegisterIndex = 0, // unused
.ValueSize = ValueSize,
// NOTE: Cache serialization will subtract the unit entry address later
.SiteAddress = SiteAddress},
},
};
}
void Arm64JITCore::InsertGuestPatchableMove(FEXCore::CPU::RelocationTypes Type, ARMEmitter::Register Reg, uint64_t Value,
uint64_t SiteAddress, uint8_t ValueSize) {
Relocation MoveABI = Relocation::Default();
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = Type};
MoveABI.GuestPatchableData.RegisterIndex = Reg.Idx();
MoveABI.GuestPatchableData.ValueSize = ValueSize;
MoveABI.GuestPatchableData.SiteAddress = SiteAddress;
// this might get patched on disk cache load
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Value, FEXCore::CPU::Arm64Emitter::PadType::DOPAD);
Relocations.emplace_back(MoveABI);
}
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations(uint64_t GuestBaseAddress) {
// Rebase relocations to library base address
for (auto& Relocation : Relocations) {
@@ -342,8 +342,8 @@ DEF_OP(TelemetrySetValue) {
(void)Bind(&LoopTop);
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP4, TMP3, TMP2);
(void)cbnz(ARMEmitter::Size::i32Bit, TMP4, &LoopTop);
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
}
#endif
}
+117 -83
View File
@@ -55,40 +55,12 @@ DEF_OP(ExitFunction) {
uint64_t NewRIP;
if constexpr (Context::BLOCK_DEBUGGING) {
// Skip block linking when BLOCK_DEBUGGING as it adds overhead and is unncessary.
// This is a debug only feature and doesn't need caching help.
bool IsInlineRIP = IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP);
ARMEmitter::ForwardLabel l_ExitLink;
if (IsInlineRIP) {
ldr(TMP1, &l_ExitLink);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
} else {
auto RipReg = GetReg(Op->NewRIP);
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
}
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
br(TMP2);
if (IsInlineRIP) {
BindOrRestart(&l_ExitLink);
dc64(NewRIP);
}
return;
}
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
#ifdef ARCHITECTURE_arm64ec
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
if (Op->PatchSiteAddress) {
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE, EC_CALL_CHECKER_PC_REG, NewRIP, Op->PatchSiteAddress,
Op->PatchSiteSize);
} else {
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
}
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionEC));
br(TMP2);
} else {
@@ -178,7 +150,6 @@ DEF_OP(ExitFunction) {
ARMEmitter::ForwardLabel TFUnset;
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
// todo do we need to account for cache patching here?
InsertGuestRIPMove(TMP1, NewRIP);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
@@ -186,7 +157,7 @@ DEF_OP(ExitFunction) {
(void)Bind(&TFUnset);
}
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call, Op->PatchSiteAddress, Op->PatchSiteSize);
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
(void)Bind(&l_CallReturn);
#ifdef ARCHITECTURE_arm64ec
}
@@ -283,11 +254,74 @@ DEF_OP(CondJump) {
}
DEF_OP(Syscall) {
SpillStaticRegs(TMP1);
auto Op = IROp->C<IR::IROp_Syscall>();
// Arguments are passed as follows:
// X0: SyscallHandler
// X1: ThreadState
// X2: Pointer to SyscallArguments
// Jump to the syscall dispatch handler. We won't return after this.
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadDispatchSyscallHandler));
br(TMP1);
PushDynamicRegs(TMP1);
uint32_t GPRSpillMask = ~0U;
uint32_t FPRSpillMask = ~0U;
SpillStaticRegs(TMP1, {
.GPRSpillMask = GPRSpillMask,
.FPRSpillMask = FPRSpillMask,
});
// Now that we are spilled, store in the state that we are in a syscall
// Still without overwriting registers that matter
// 16bit LoadConstant to be a single instruction
// This gives the signal handler a value to check to see if we are in a syscall at all
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
continue;
}
str(GetReg(Op->Header.Args[i]).X(), ARMEmitter::Reg::rsp, i * 8);
}
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerObj));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerFunc));
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
// SP supporting move
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
} else {
blr(ARMEmitter::Reg::r3);
}
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
// Result is now in x0
// Fix the stack and any values that were stepped on
FillStaticRegs({
.OptionalReg = ARMEmitter::Reg::r1,
.OptionalReg2 = ARMEmitter::Reg::r2,
.GPRFillMask = GPRSpillMask,
.FPRFillMask = FPRSpillMask,
});
// Now the registers we've spilled are back in their original host registers
// We can safely claim we are no longer in a syscall
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
PopDynamicRegs();
const auto OSABI = CTX->SyscallHandler->GetOSABI();
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
// Move result to its destination register.
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
}
}
DEF_OP(Thunk) {
@@ -322,74 +356,74 @@ DEF_OP(Thunk) {
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
auto Base = GetReg(Op->Address).X();
auto OldCode = Op->CodeOriginal.data();
auto Base = GetReg(Op->Header.Args[0]).X();
int len = Op->CodeLength;
int Offset = 0;
ARMEmitter::ForwardLabel Fail;
const auto Dst = GetReg(Node);
const auto CRC32Reg = GetReg(Op->crc);
// Changes to TMP1
auto WorkingReg = ARMEmitter::XReg::zr;
auto BaseReg = TMP2;
auto TmpDataReg = TMP3;
mov(ARMEmitter::Size::i64Bit, BaseReg, Base);
auto EmitCheck = [&](size_t Size, auto&& LoadData) {
while (len >= Size) {
LoadData();
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
len -= Size;
Offset += Size;
}
};
while (len >= 8) {
ldr<ARMEmitter::IndexType::POST>(TmpDataReg, BaseReg, 8);
crc32x(TMP1, WorkingReg, TmpDataReg);
len -= 8;
WorkingReg = TMP1;
}
EmitCheck(8, [&]() {
ldr(TMP1, Base, Offset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset));
});
while (len >= 4) {
ldr<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 4);
crc32w(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
len -= 4;
WorkingReg = TMP1;
}
EmitCheck(4, [&]() {
ldr(TMP1.W(), Base, Offset);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset));
});
while (len >= 2) {
ldrh<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 2);
crc32h(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
len -= 2;
WorkingReg = TMP1;
}
EmitCheck(2, [&]() {
ldrh(TMP1.W(), Base, Offset);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset));
});
while (len >= 1) {
ldrb<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 1);
crc32b(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
len -= 1;
WorkingReg = TMP1;
}
sub(ARMEmitter::Size::i32Bit, Dst, TMP1, CRC32Reg);
EmitCheck(1, [&]() {
ldrb(TMP1.W(), Base, Offset);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset));
});
ARMEmitter::ForwardLabel End;
cbz_OrRestart(ARMEmitter::Size::i32Bit, Dst, &End);
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
b_OrRestart(&End);
BindOrRestart(&Fail);
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
BindOrRestart(&End);
}
DEF_OP(ThreadRemoveCodeEntry) {
auto Op = IROp->C<IR::IROp_ThreadRemoveCodeEntry>();
SpillStaticRegs(TMP1);
// Store the new RIP to go to.
str(GetReg(Op->NewRIP).X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
// Move the entry to ABI before saving state.
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->EntryToInvalidate));
PushDynamicRegs(TMP4);
SpillStaticRegs(TMP4);
// Arguments are passed as follows:
// X0: Thread
// X1: RIPToInvalidate
// X1: RIP
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
// Jump to the invalidate dispatch handler. We won't return after this.
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadDispatchRemoveCodeEntry));
br(ARMEmitter::XReg::x2);
// TODO: Relocations don't seem to be wired up to this...?
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry, CPU::Arm64Emitter::PadType::AUTOPAD);
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadRemoveCodeEntryFromJIT));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
blr(ARMEmitter::Reg::r2);
}
FillStaticRegs();
// Fix the stack and any values that were stepped on
PopDynamicRegs();
}
DEF_OP(CPUID) {
@@ -292,6 +292,8 @@ DEF_OP(Vector_FToS) {
frinti(SubEmitSize, Dst.Z(), Mask.Merging(), Vector.Z());
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Dst.Z(), SubEmitSize);
} else {
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector);
if (OpSize == IR::OpSize::i64Bit) {
frinti(SubEmitSize, Dst.D(), Vector.D());
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
@@ -557,30 +559,26 @@ DEF_OP(Vector_F64ToI32) {
}
}
} else {
// This has a known precision issue that isn't easily resolvable without throwing away performance.
// Doing the conversion in multi-stage steps has an issue that you can lose precision in the f32->i32 step if your source was f64.
// To get around this with ASIMD FEX needs to use fcvtzs (Scalar, Integer, to GPR) for each F64 to be directly converted to i32.
// This is a very costly transform that the SVE path doesn't need to do since it supports f64->i32 directly.
// If this precision issue is necessary then we can add an option for it in the future.
///< Round float to integral depending on rounding mode.
///< skip TowardsZero as fcvtzs below already truncates toward zero on its own
auto CVTReg = Dst.Q();
switch (Round) {
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::TowardsZero: CVTReg = Vector.Q(); break;
case IR::RoundMode::TowardsZero: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
}
///< Convert f64 directly to i64
fcvtzs(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), CVTReg);
// Now narrow from f64 to f32.
fcvtn(ARMEmitter::SubRegSize::i32Bit, Dst.Q(), Dst.Q());
///< Saturating narrow i64 -> i32
///
///< The caller(Vector_CVT_Float_To_Int32Impl) only fixes up positive overflow:
///< it tests MaxF > Src (MaxF = 2^31) and swaps in CVTMAX_I32 (0x80000000) where
///< the test fails.
///
///< Sources below INT32_MIN are handled by sqxtn:
///< ARM saturates to INT32_MIN, which is 0x80000000 the same value as
///< x86's integer-indefinite value.
sqxtn(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Dst.D());
///< Convert the two F32 integrals to real integers.
fcvtzs(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Dst.D());
}
}
@@ -324,62 +324,24 @@ DEF_OP(PCLMUL) {
const auto Op = IROp->C<IR::IROp_PCLMUL>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Src1 = GetVReg(Op->Src1);
const auto Src2 = GetVReg(Op->Src2);
if (HostSupportsSVE256 && Is256Bit) {
switch (Op->Selector) {
case 0b00000000: {
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), Src2.Z());
break;
}
case 0b00000001: {
trn2(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), Src1.Z(), Src1.Z());
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), VTMP1.Z(), Src2.Z());
break;
}
case 0b00010000:
trn2(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), Src2.Z(), Src2.Z());
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), VTMP1.Z());
break;
case 0b00010001: {
pmullt(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), Src2.Z());
break;
}
default: {
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
break;
}
}
} else {
switch (Op->Selector) {
case 0b00000000: {
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D());
break;
}
case 0b00000001: {
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
break;
}
case 0b00010000: {
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
break;
}
case 0b00010001: {
pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q());
break;
}
default: {
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
break;
}
}
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
switch (Op->Selector) {
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
case 0b00000001:
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
break;
case 0b00010000:
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
break;
case 0b00010001: pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q()); break;
default: LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector); break;
}
}
+69 -130
View File
@@ -36,14 +36,7 @@ $end_info$
#include <cstdio>
#include <cstring>
#include <optional>
#include <type_traits>
#include <unistd.h>
#ifdef _WIN32
#include <atomic>
#else
#include <FEXHeaderUtils/Syscalls.h>
#endif
namespace {
struct DivRem {
@@ -70,36 +63,8 @@ LDIV(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) {
};
}
#ifndef _WIN32
static std::optional<uint64_t>
RDRANDFallback(uint64_t Reseed) {
uint64_t Value {};
FHU::Syscalls::getrandom(&Value, sizeof(Value), 0);
return Value;
}
#else
// Windows does not have an equivalent to getrandom() without dynamically linking
// bcrypt et al. Since this fallback is just for compat it does not need to be
// cryptographic, so instead we vendor SplitMix64 as a naive fallback.
// Reference implementation by Sebastiano Vigna, public domain (CC0)
// https://prng.di.unimi.it/splitmix64.c
static std::atomic<uint64_t>
RNGState {static_cast<uint64_t>(__builtin_readcyclecounter())};
static std::optional<uint64_t> RDRANDFallback(uint64_t Reseed) {
const uint64_t State = RNGState.load(std::memory_order_relaxed) + 0x9E3779B97F4A7C15ULL;
RNGState.store(State, std::memory_order_relaxed);
uint64_t Value = (State ^ (State >> 30)) * 0xBF58476D1CE4E5B9ULL;
Value = (Value ^ (Value >> 27)) * 0x94D049BB133111EBULL;
return Value ^ (Value >> 31);
}
#endif
static void PrintValue(uint64_t Value) {
static void
PrintValue(uint64_t Value) {
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
}
@@ -651,13 +616,13 @@ void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::Ref Node) {}
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
: CPUBackend(*ctx, Thread)
, Arm64Emitter(ctx)
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128 != 0}
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256 != 0}
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128}
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256}
, HostSupportsAVX256 {ctx->HostFeatures.SupportsAVX && ctx->HostFeatures.SupportsSVE256}
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES != 0}
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP != 0}
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES}
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP}
, CTX {ctx}
, TempCodeBufferAllocator(ctx->CPUBackendAllocator, 0) {
, TempAllocator(ctx->CPUBackendAllocator, 0) {
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
@@ -665,7 +630,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
RAPass->AddRegisters(IR::RegClass::GPRFixed, StaticRegisters.size());
RAPass->AddRegisters(IR::RegClass::FPR, GeneralFPRegisters.size());
RAPass->AddRegisters(IR::RegClass::FPRFixed, StaticFPRegisters.size());
RAPass->SetNumPairRegs(PairRegisters);
RAPass->PairRegs = PairRegisters;
{
// Set up pointers that the JIT needs to load
@@ -699,20 +664,28 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
Ptrs.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore::ExitFunctionLink);
Ptrs.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
Ptrs.LDIV = reinterpret_cast<uint64_t>(LDIV);
Ptrs.RDRANDFallback = reinterpret_cast<uint64_t>(RDRANDFallback);
}
CurrentCodeBuffer = SharedCodeBuffers.GetLatest();
CurrentCodeBuffer = CodeBuffers.GetLatest();
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
}
void Arm64JITCore::EmitDetectionString() {
const char JITString[] = "FEXJIT::Arm64JITCore::";
EmitString(JITString);
Align();
}
void Arm64JITCore::ClearCache() {
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
auto PrevCodeBuffer = CurrentCodeBuffer;
auto lk = PrevCodeBuffer->LookupCache->AcquireWriteLock();
auto CodeBuffer = AcquireNewSharedCodeBuffer();
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CodeBuffer->LookupCache, lk);
auto CodeBuffer = GetEmptyCodeBuffer();
SetBuffer(CodeBuffer->Ptr, CodeBuffer->AllocatedSize);
EmitDetectionString();
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache, lk);
}
Arm64JITCore::~Arm64JITCore() {}
@@ -802,19 +775,12 @@ void Arm64JITCore::EmitTFCheck() {
void Arm64JITCore::EmitSuspendInterruptCheck() {
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
// Trigger a fault if there are any pending interrupts
// Used only for gdbserver at the moment
constexpr size_t InterruptPageOffset =
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState);
if constexpr (InterruptPageOffset <= 32760) {
str(ARMEmitter::XReg::zr, STATE, InterruptPageOffset);
} else {
// Need to use vector 128-bit store for this range.
// Doesn't matter which register we use to store.
str(ARMEmitter::QReg::q0, STATE, InterruptPageOffset);
}
// Used only for suspend on WIN32 at the moment
strb(ARMEmitter::XReg::zr, STATE,
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
}
#ifdef _WIN32
#ifdef ARCHITECTURE_arm64ec
static constexpr uint16_t SuspendMagic {0xCAFE};
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
@@ -849,33 +815,6 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
EmitSuspendInterruptCheck();
}
CodeBuffer::CodeBufferAllocation Arm64JITCore::AllocateCodeBufferInSharedCache(size_t Size) {
CodeBuffer::CodeBufferAllocation AllocatedInfo {};
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
"doesn't match up!\n");
// Bring CodeBuffer up to date
if (auto Prev = CheckCodeBufferUpdate()) {
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
auto lk = ThreadState->LookupCache->AcquireWriteLock();
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
}
// Attempt to allocate a buffer from the SharedCodeBuffers.
while (AllocatedInfo.BufferAllocationOffset == nullptr) {
AllocatedInfo = CurrentCodeBuffer->AtomicAllocateBuffer(Size);
if (AllocatedInfo.BufferAllocationOffset == nullptr) {
// If it didn't fit then clear the buffer and try again.
// This has the possibility of migrating the SharedCodeBuffer. See above in `Arm64JITCore::ClearCache()`
CTX->ClearCodeCache(ThreadState);
continue;
}
}
return AllocatedInfo;
}
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
@@ -897,7 +836,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
case RestartOptions::Control::NeedsLargerJITSpace:
// Get rid of the claimed buffer immediately, we can't fit in it at all.
TempCodeBufferAllocator.UnclaimBuffer();
TempAllocator.UnclaimBuffer();
SSANodeMultiplier *= 2;
break;
default: LOGMAN_MSG_A_FMT("Unhandled Arm64 restart condition!");
@@ -918,7 +857,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
// This minimizes lock contention of CodeBufferWriteMutex.
auto TempCodeBufferInfo = TempCodeBufferAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
auto TempCodeBufferInfo = TempAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
auto TempCodeBuffer = TempCodeBufferInfo.Ptr;
const uint32_t UsableBufferRange = TempCodeBufferInfo.Size - FEXCore::Utils::FEX_PAGE_SIZE;
@@ -928,7 +867,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
ThreadState->JITGuardOverflowArgument = FEXCore::ToUnderlying(RestartOptions::Control::NeedsLargerJITSpace);
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
LOGMAN_THROW_A_FMT(GetCursorOffset() == 0, "Needs to be zero");
// Put the code header at the start of the data block.
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
@@ -964,6 +902,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
PendingCallReturnTargetLabel = nullptr;
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
using namespace FEXCore::IR;
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
@@ -1055,15 +994,9 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// This is a ExitFunctionLinkData struct
BindOrRestart(&l_ExitLink);
dc64(0); // HostCode
if (PendingJumpThunk.PatchSiteAddress) {
// GuestRIP with an extra step
PlaceNamedSymbolLiteral(
InsertGuestPatchableRIPLiteral(PendingJumpThunk.GuestRIP, PendingJumpThunk.PatchSiteAddress, PendingJumpThunk.PatchSiteSize));
} else {
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
}
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
dc64(0); // HostCode
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
}
BindOrRestart(&l_ExitLink);
@@ -1123,47 +1056,69 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
}
SetCursorOffset(JITRIPEntriesLocation - CodeData.BlockBegin);
// Make sure code is 16B aligned on the tail.
// Can't use Align16B here as vl64pair can cause non-4byte alignment.
Align(16);
Align();
// Beginning of emission is guaranteed to be offset zero. So the code data size is just the current cursor offset.
CodeData.Size = GetCursorOffset();
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
// Finalize and write block tail data
JITBlockTail.Size = CodeData.Size;
{
auto PrevCur = GetCursorOffset();
memcpy(JITBlockTailLocation, &JITBlockTail, sizeof(JITBlockTail));
SetCursorOffset(JITBlockTailLocation - CodeData.BlockBegin + offsetof(JITCodeTail, RIP));
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(JITBlockTail.RIP));
// Emitter buffer is no longer used, guard against misuse by setting to nullptr.
SetBuffer(nullptr, 0);
SetCursorOffset(PrevCur);
}
// Migrate the compile output from temporary storage to the actual CodeBuffer.
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
{
LOGMAN_THROW_A_FMT(CodeData.Size % 16 == 0, "Needs to be 16B aligned!");
auto CodeBufferLock = std::unique_lock {CodeBuffers.CodeBufferWriteMutex};
auto AllocatedInfo = AllocateCodeBufferInSharedCache(CodeData.Size);
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
LOGMAN_THROW_A_FMT((reinterpret_cast<uintptr_t>(AllocatedInfo.BufferAllocationOffset) % 16) == 0, "Allocated buffer wasn't 16B "
"aligned?");
// Query size of generated code
const auto TempSize = GetCursorOffset();
// Bring CodeBuffer up to date
{
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
"doesn't match up!\n");
if (auto Prev = CheckCodeBufferUpdate()) {
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
auto lk = ThreadState->LookupCache->AcquireWriteLock();
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
}
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
SetBuffer(CurrentCodeBuffer->Ptr, CurrentCodeBuffer->AllocatedSize);
SetCursorOffset(CodeBuffers.LatestOffset);
Align16B();
if ((GetCursorOffset() + TempSize) > CurrentCodeBuffer->UsableSize()) {
CTX->ClearCodeCache(ThreadState);
}
CodeBuffers.LatestOffset = GetCursorOffset();
}
// Adjust host addresses
const auto Delta = AllocatedInfo.BufferAllocationOffset - CodeData.BlockBegin;
const auto Delta = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
CodeData.BlockBegin += Delta;
for (auto& EntryPoint : CodeData.EntryPoints) {
EntryPoint.second += Delta;
}
CodeBegin += Delta;
CodeData.HostCodeOffset = CodeData.BlockBegin - CurrentCodeBuffer->GetBufferBase();
for (std::size_t Idx = PrevNumAllocations; Idx != Relocations.size(); ++Idx) {
Relocations[Idx].Header.Offset += CodeBuffers.LatestOffset;
}
// Copy over CodeBuffer contents
memcpy(AllocatedInfo.BufferAllocationOffset, TempCodeBuffer, CodeData.Size);
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
SetCursorOffset(CodeBuffers.LatestOffset + TempSize);
CodeBuffers.LatestOffset = GetCursorOffset();
}
TempCodeBufferAllocator.DelayedDisownBuffer();
TempAllocator.DelayedDisownBuffer();
ClearICache(CodeBegin, CodeOnlySize);
@@ -1199,22 +1154,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
return std::move(CodeData);
}
CPUBackend::CompiledCode Arm64JITCore::LoadCachedCode(std::span<const uint8_t> HostBytes) {
// we stored it aligned, better still be?
LOGMAN_THROW_A_FMT(HostBytes.size() % 16 == 0, "Needs to be 16B aligned!");
auto AllocatedInfo = AllocateCodeBufferInSharedCache(HostBytes.size());
uint8_t* Dest = AllocatedInfo.BufferAllocationOffset;
memcpy(Dest, HostBytes.data(), HostBytes.size());
ClearICache(Dest, HostBytes.size());
CPUBackend::CompiledCode Result;
Result.BlockBegin = Dest;
Result.Size = HostBytes.size();
Result.HostCodeOffset = Dest - CurrentCodeBuffer->GetBufferBase();
return Result;
}
void Arm64JITCore::ResetStack() {
if (SpillSlots == 0) {
return;
+5 -18
View File
@@ -54,9 +54,6 @@ public:
CPUBackend::CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) override;
[[nodiscard]]
CPUBackend::CompiledCode LoadCachedCode(std::span<const uint8_t> HostBytes) override;
void ClearCache() override;
void ClearRelocations() override {
@@ -105,12 +102,10 @@ private:
uint64_t CallerAddress;
uint64_t GuestRIP;
ARMEmitter::ForwardLabel Label;
uint64_t PatchSiteAddress = 0;
uint8_t PatchSiteSize = 0;
};
fextl::vector<PendingJumpThunk> PendingJumpThunks;
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempCodeBufferAllocator;
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempAllocator;
static uint64_t ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
@@ -347,8 +342,8 @@ private:
uint32_t End;
};
void EmitLinkedBranch(uint64_t GuestRIP, bool Call, uint64_t PatchSiteAddress = 0, uint8_t PatchSiteSize = 0) {
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}, PatchSiteAddress, PatchSiteSize});
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
auto& Thunk = PendingJumpThunks.back();
BindOrRestart(&Thunk.Label);
if (Call) {
@@ -531,6 +526,8 @@ private:
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
IR::RegisterAllocationPass* RAPass {};
FEXCore::Core::DebugData* DebugData {};
@@ -564,9 +561,6 @@ private:
*/
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
void InsertGuestPatchableMove(FEXCore::CPU::RelocationTypes Type, ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress,
uint8_t ValueSize);
/**
* @brief Inserts a named symbol as a literal in memory
*
@@ -586,11 +580,6 @@ private:
*/
NamedSymbolLiteralPair InsertGuestRIPLiteral(uint64_t GuestRIP);
/**
* @brief Like InsertGuestRIPLiteral, but with patch information to recompute value from live guest bytes at cache load time
*/
NamedSymbolLiteralPair InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize);
/**
* @brief Place the named symbol literal relocation in memory
*
@@ -637,8 +626,6 @@ private:
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
[[nodiscard]] CodeBuffer::CodeBufferAllocation AllocateCodeBufferInSharedCache(size_t Size);
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
///< Unhandled handler
+11 -13
View File
@@ -267,7 +267,7 @@ DEF_OP(LoadContextIndexed) {
ldr(Dst.Q(), TMP1, Op->BaseOffset);
} else {
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, Op->BaseOffset);
ldur(Dst.Q(), TMP1);
ldur(Dst.Q(), TMP1, Op->BaseOffset);
}
break;
case IR::OpSize::i256Bit:
@@ -333,7 +333,7 @@ DEF_OP(StoreContextIndexed) {
str(Value.Q(), TMP1, Op->BaseOffset);
} else {
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, Op->BaseOffset);
stur(Value.Q(), TMP1);
stur(Value.Q(), TMP1, Op->BaseOffset);
}
break;
case IR::OpSize::i256Bit:
@@ -568,7 +568,7 @@ DEF_OP(LoadDF) {
DEF_OP(ContextClear) {
auto Op = IROp->C<IR::IROp_ContextClear>();
if (CTX->HostFeatures.PreferZVAForVZero) {
if (CTX->HostFeatures.SupportsCLZERO) {
// We can use CLZero directly when hardware supports it.
// Provides a fairly generous speed-up on Ampere1A hardware.
// TODO: When FEAT_MOPS hardware ships, test memset using MOPS.
@@ -2400,14 +2400,13 @@ DEF_OP(CacheLineClear) {
// Clear dcache only
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
if (CTX->HostFeatures.DCacheSize() >= 64U) {
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
dc(ARMEmitter::DataCacheOperation::CIVAC, MemReg);
} else {
auto CurrentWorkingReg = MemReg.X();
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheSize()); ++i) {
dc(ARMEmitter::DataCacheOperation::CIVAC, CurrentWorkingReg);
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheSize());
for (size_t i = 0; i < std::max(1U, CTX->HostFeatures.DCacheLineSize / 64U); ++i) {
dc(ARMEmitter::DataCacheOperation::CIVAC, TMP1);
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
CurrentWorkingReg = TMP1;
}
}
@@ -2429,14 +2428,13 @@ DEF_OP(CacheLineClean) {
auto MemReg = GetReg(Op->Addr);
// Clean dcache only
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
if (CTX->HostFeatures.DCacheSize() >= 64U) {
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
dc(ARMEmitter::DataCacheOperation::CVAC, MemReg);
} else {
auto CurrentWorkingReg = MemReg.X();
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheSize()); ++i) {
dc(ARMEmitter::DataCacheOperation::CVAC, CurrentWorkingReg);
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheSize());
for (size_t i = 0; i < std::max(1U, CTX->HostFeatures.DCacheLineSize / 64U); ++i) {
dc(ARMEmitter::DataCacheOperation::CVAC, TMP1);
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
CurrentWorkingReg = TMP1;
}
}
+5 -37
View File
@@ -73,8 +73,8 @@ DEF_OP(Break) {
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant);
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
switch (Op->Reason.Signal) {
case Core::FAULT_SIGILL:
@@ -168,7 +168,7 @@ DEF_OP(PushRoundingMode) {
} else {
LOGMAN_THROW_A_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(3 << 22));
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(Op->RoundMode << 22));
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, (Op->RoundMode == 2 ? 1 : 2) << 22);
}
@@ -282,7 +282,7 @@ DEF_OP(ProcessorID) {
// Load the values returned by the kernel
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::WReg::w0, ARMEmitter::WReg::w1, ARMEmitter::Reg::rsp);
// Deallocate stack space
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
// Now that we are done in the syscall we need to carefully peel back the state
// First unspill the registers from before
@@ -309,39 +309,7 @@ DEF_OP(ProcessorID) {
DEF_OP(RDRAND) {
auto Op = IROp->C<IR::IROp_RDRAND>();
if (CTX->HostFeatures.SupportsRAND) {
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
return;
}
// Software fallback, call the host RNG generator.
PushDynamicRegs(TMP4);
SpillStaticRegs(TMP4);
// x0 = Reseed
// x1 = Generator
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Op->GetReseeded ? 1 : 0);
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.RDRANDFallback));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<__uint128_t, uint64_t>(ARMEmitter::Reg::r1);
} else {
blr(ARMEmitter::Reg::r1);
}
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
mov(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Reg::r1);
}
FillStaticRegs();
PopDynamicRegs();
// Results are in x0, x1
// std::optional<uint64_t>: value in x0, engaged flag in the low byte of x1. Match the hardware behaviour of setting Z when
// no number was produced.
mov(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1);
tst(ARMEmitter::Size::i64Bit, TMP2, 0xFF);
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
}
DEF_OP(Yield) {
@@ -25,22 +25,6 @@ enum class RelocationTypes : uint32_t {
// 4 instruction constant generation
// Aligned to struct RelocGuestRIP
RELOC_GUEST_RIP_MOVE,
// The frontend flagged those regions as patchable by the disk cache
// Aligned to struct RelocGuestPatchableData
RELOC_GUEST_PATCHABLE_DATA_MOVE,
// Same as GuestRipLiteral but patchable
// Aligned to struct RelocGuestPatchableData
RELOC_GUEST_PATCHABLE_RIP_LITERAL,
// Like PATCHABLE_RIP_LITERAL but puts it in a register
// Aligned to struct RelocGuestPatchableData
RELOC_GUEST_PATCHABLE_RIP_MOVE,
// Patchable guest CRC
// Aligned to struct RelocGuestPatchableData
RELOC_GUEST_PATCHABLE_CRC_MOVE,
};
struct FEX_PACKED RelocationHeader final {
@@ -89,20 +73,6 @@ struct RelocGuestRIP final {
uint32_t pad2[6] {};
};
struct RelocGuestPatchableData final {
RelocationHeader Header {};
uint8_t RegisterIndex;
uint8_t ValueSize;
char Pad[2];
uint64_t SiteAddress;
uint32_t pad2[6] {};
};
union Relocation {
// Clang 16 Can't default-initialize this union
static Relocation Default() {
@@ -123,8 +93,6 @@ union Relocation {
RelocNamedThunkMove NamedThunkMove;
RelocGuestRIP GuestRIP;
RelocGuestPatchableData GuestPatchableData;
};
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl&, RelocNamedSymbolLiteral::NamedSymbol);
+169 -556
View File
@@ -1036,28 +1036,24 @@ DEF_OP(VMov) {
const auto Dst = GetVReg(Node);
const auto Source = GetVReg(Op->Source);
const auto Sub64BitHandler = [&](ARMEmitter::SubRegSize InsertSize) {
if (Dst != Source) {
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
ins(InsertSize, Dst, 0, Source, 0);
} else {
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
ins(InsertSize, VTMP1, 0, Source, 0);
mov(Dst.Q(), VTMP1.Q());
}
};
switch (OpSize) {
case IR::OpSize::i8Bit: {
Sub64BitHandler(ARMEmitter::SubRegSize::i8Bit);
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
ins(ARMEmitter::SubRegSize::i8Bit, VTMP1, 0, Source, 0);
mov(Dst.Q(), VTMP1.Q());
break;
}
case IR::OpSize::i16Bit: {
Sub64BitHandler(ARMEmitter::SubRegSize::i16Bit);
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 0, Source, 0);
mov(Dst.Q(), VTMP1.Q());
break;
}
case IR::OpSize::i32Bit: {
Sub64BitHandler(ARMEmitter::SubRegSize::i32Bit);
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
ins(ARMEmitter::SubRegSize::i32Bit, VTMP1, 0, Source, 0);
mov(Dst.Q(), VTMP1.Q());
break;
}
case IR::OpSize::i64Bit: {
@@ -1099,23 +1095,16 @@ DEF_OP(VAddP) {
if (HostSupportsSVE256 && Is256Bit) {
const auto Pred = PRED_TMP_32B.Merging();
// SVE ADDP is a destructive operation, so we need a temporary if
// the destination and the lower vector don't alias.
auto LHS = Dst;
if (Dst != VectorLower) {
movprfx(VTMP1.Z(), VectorLower.Z());
LHS = VTMP1;
}
// SVE ADDP is a destructive operation, so we need a temporary
movprfx(VTMP1.Z(), VectorLower.Z());
// Unlike Adv. SIMD's version of ADDP, which acts like it concats the
// upper vector onto the end of the lower vector and then performs
// pairwise addition, the SVE version actually interleaves the
// results of the pairwise addition (gross!), so we need to undo that.
addp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
// Extract the upper half first, since Dst may alias the LHS.
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
addp(SubRegSize, VTMP1.Z(), Pred, VTMP1.Z(), VectorUpper.Z());
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP1.Z());
uzp2(SubRegSize, VTMP2.Z(), VTMP1.Z(), VTMP1.Z());
// Merge upper half with lower half.
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
@@ -1141,16 +1130,9 @@ DEF_OP(VOrn) {
const auto Vector2 = GetVReg(Op->Vector2);
if (HostSupportsSVE256 && Is256Bit) {
if (Dst == Vector1) {
bsl2n(Dst.Z(), Dst.Z(), Vector2.Z(), Dst.Z());
} else if (Dst == Vector2) {
const auto Pred = PRED_TMP_32B.Merging();
not_(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Pred, Dst.Z());
orr(Dst.Z(), Vector1.Z(), Dst.Z());
} else {
movprfx(Dst.Z(), Vector1.Z());
bsl2n(Dst.Z(), Dst.Z(), Vector2.Z(), Vector1.Z());
}
const auto Pred = PRED_TMP_32B.Merging();
not_(ARMEmitter::SubRegSize::i8Bit, VTMP1.Z(), Pred, Vector2.Z());
orr(Dst.Z(), Vector1.Z(), VTMP1.Z());
} else if (Is128Bit) {
orn(Dst.Q(), Vector1.Q(), Vector2.Q());
} else {
@@ -1174,7 +1156,8 @@ DEF_OP(VFAddV) {
if (HostSupportsSVE256 && Is256Bit) {
const auto Pred = PRED_TMP_32B.Merging();
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
} else if (HostSupportsSVE128) {
}
if (HostSupportsSVE128) {
const auto Pred = PRED_TMP_16B.Merging();
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
} else {
@@ -1201,16 +1184,20 @@ DEF_OP(VAddV) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
if (ElementSize == IR::OpSize::i64Bit) {
const auto Mask = PRED_TMP_32B.Zeroing();
uaddv(SubRegSize.Vector, Dst.D(), Mask, Vector.Z());
} else {
const auto Mask = ARMEmitter::PReg::p0;
uaddv(SubRegSize.Vector, VTMP1.D(), Mask, Vector.Z());
mov_imm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), 0);
ptrue(SubRegSize.Vector, Mask, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Mask.Merging(), VTMP1.Z());
}
// SVE doesn't have an equivalent ADDV instruction, so we make do
// by performing two Adv. SIMD ADDV operations on the high and low
// 128-bit lanes and then sum them up.
const auto Mask = PRED_TMP_32B.Zeroing();
const auto CompactPred = ARMEmitter::PReg::p0;
// Select all our upper elements to run ADDV over them.
not_(CompactPred, Mask, PRED_TMP_16B);
compact(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), CompactPred, Vector.Z());
addv(SubRegSize.Vector, VTMP2.Q(), Vector.Q());
addv(SubRegSize.Vector, VTMP1.Q(), VTMP1.Q());
add(SubRegSize.Vector, Dst.Q(), VTMP1.Q(), VTMP2.Q());
} else {
if (ElementSize == IR::OpSize::i64Bit) {
addp(SubRegSize.Scalar, Dst, Vector);
@@ -1299,7 +1286,6 @@ DEF_OP(VFAddP) {
const auto Op = IROp->C<IR::IROp_VFAddP>();
const auto OpSize = IROp->Size;
const auto IsScalar = OpSize == IR::OpSize::i64Bit;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
@@ -1312,26 +1298,19 @@ DEF_OP(VFAddP) {
if (HostSupportsSVE256 && Is256Bit) {
const auto Pred = PRED_TMP_32B.Merging();
// SVE FADDP is a destructive operation, so we need a temporary if
// the destination and the lower vector don't alias.
auto LHS = Dst;
if (Dst != VectorLower) {
movprfx(VTMP1.Z(), VectorLower.Z());
LHS = VTMP1;
}
// SVE FADDP is a destructive operation, so we need a temporary
movprfx(VTMP1.Z(), VectorLower.Z());
// Unlike Adv. SIMD's version of FADDP, which acts like it concats the
// upper vector onto the end of the lower vector and then performs
// pairwise addition, the SVE version actually interleaves the
// results of the pairwise addition (gross!), so we need to undo that.
faddp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
faddp(SubRegSize, VTMP1.Z(), Pred, VTMP1.Z(), VectorUpper.Z());
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP1.Z());
uzp2(SubRegSize, VTMP2.Z(), VTMP1.Z(), VTMP1.Z());
// Merge upper half with lower half.
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
} else if (IsScalar) {
faddp(SubRegSize, Dst.D(), VectorLower.D(), VectorUpper.D());
} else {
faddp(SubRegSize, Dst.Q(), VectorLower.Q(), VectorUpper.Q());
}
@@ -1547,14 +1526,9 @@ DEF_OP(VFRecp) {
return;
}
if (Dst != Vector) {
fmov(SubRegSize.Vector, Dst.Z(), 1.0);
fdiv(SubRegSize.Vector, Dst.Z(), Pred, Dst.Z(), Vector.Z());
} else {
fmov(SubRegSize.Vector, VTMP1.Z(), 1.0);
fdiv(SubRegSize.Vector, VTMP1.Z(), Pred, VTMP1.Z(), Vector.Z());
mov(Dst.Z(), VTMP1.Z());
}
fmov(SubRegSize.Vector, VTMP1.Z(), 1.0);
fdiv(SubRegSize.Vector, VTMP1.Z(), Pred, VTMP1.Z(), Vector.Z());
mov(Dst.Z(), VTMP1.Z());
} else {
if (IsScalar) {
if (ElementSize == IR::OpSize::i32Bit && HostSupportsRPRES) {
@@ -1806,14 +1780,10 @@ DEF_OP(VUMin) {
break;
}
case IR::OpSize::i64Bit: {
if (Dst != Vector1 && Dst != Vector2) {
cmhi(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
bsl(Dst.Q(), Vector2.Q(), Vector1.Q());
} else {
cmhi(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
bsl(VTMP1.Q(), Vector2.Q(), Vector1.Q());
mov(Dst.Q(), VTMP1.Q());
}
cmhi(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
mov(VTMP2.Q(), Vector1.Q());
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
mov(Dst.Q(), VTMP2.Q());
break;
}
default: break;
@@ -1859,14 +1829,10 @@ DEF_OP(VSMin) {
break;
}
case IR::OpSize::i64Bit: {
if (Dst != Vector1 && Dst != Vector2) {
cmgt(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
bsl(Dst.Q(), Vector2.Q(), Vector1.Q());
} else {
cmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
bsl(VTMP1.Q(), Vector2.Q(), Vector1.Q());
mov(Dst.Q(), VTMP1.Q());
}
cmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
mov(VTMP2.Q(), Vector1.Q());
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
mov(Dst.Q(), VTMP2.Q());
break;
}
default: break;
@@ -1912,14 +1878,10 @@ DEF_OP(VUMax) {
break;
}
case IR::OpSize::i64Bit: {
if (Dst != Vector1 && Dst != Vector2) {
cmhi(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
bsl(Dst.Q(), Vector1.Q(), Vector2.Q());
} else {
cmhi(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
bsl(VTMP1.Q(), Vector1.Q(), Vector2.Q());
mov(Dst.Q(), VTMP1.Q());
}
cmhi(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
mov(VTMP2.Q(), Vector1.Q());
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
mov(Dst.Q(), VTMP2.Q());
break;
}
default: break;
@@ -1965,14 +1927,10 @@ DEF_OP(VSMax) {
break;
}
case IR::OpSize::i64Bit: {
if (Dst != Vector1 && Dst != Vector2) {
cmgt(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
bsl(Dst.Q(), Vector1.Q(), Vector2.Q());
} else {
cmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
bsl(VTMP1.Q(), Vector1.Q(), Vector2.Q());
mov(Dst.Q(), VTMP1.Q());
}
cmgt(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
mov(VTMP2.Q(), Vector1.Q());
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
mov(Dst.Q(), VTMP2.Q());
break;
}
default: break;
@@ -2169,46 +2127,6 @@ DEF_OP(VCMPGT) {
}
}
DEF_OP(VUCMPGT) {
const auto Op = IROp->C<IR::IROp_VUCMPGT>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
const auto IsScalar = ElementSize == OpSize;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1);
const auto Vector2 = GetVReg(Op->Vector2);
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Zeroing();
const auto ComparePred = ARMEmitter::PReg::p0;
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
// General idea is to compare for unsigned greater-than, bitwise NOT
// the valid values, then ORR the NOTed values with the original
// values to form entries that are all 1s.
cmphi(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
// Restore NZCV
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
} else {
if (IsScalar) {
cmhi(SubRegSize.Scalar, Dst, Vector1, Vector2);
} else {
cmhi(SubRegSize.Vector, Dst.Q(), Vector1.Q(), Vector2.Q());
}
}
}
DEF_OP(VCMPGTZ) {
const auto Op = IROp->C<IR::IROp_VCMPGTZ>();
const auto OpSize = IROp->Size;
@@ -2832,17 +2750,17 @@ DEF_OP(VUShrSWide) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
if (ElementSize == IR::OpSize::i64Bit) {
const auto Mask = PRED_TMP_32B.Merging();
const auto Mask = PRED_TMP_32B.Merging();
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
if (ElementSize == IR::OpSize::i64Bit) {
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
} else {
lsr_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
}
} else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging();
@@ -2898,17 +2816,17 @@ DEF_OP(VSShrSWide) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
if (ElementSize == IR::OpSize::i64Bit) {
const auto Mask = PRED_TMP_32B.Merging();
const auto Mask = PRED_TMP_32B.Merging();
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
if (ElementSize == IR::OpSize::i64Bit) {
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
} else {
asr_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
}
} else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging();
@@ -2964,17 +2882,17 @@ DEF_OP(VUShlSWide) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
if (ElementSize == IR::OpSize::i64Bit) {
const auto Mask = PRED_TMP_32B.Merging();
const auto Mask = PRED_TMP_32B.Merging();
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
if (ElementSize == IR::OpSize::i64Bit) {
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
} else {
lsl_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
}
} else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging();
@@ -3062,14 +2980,9 @@ DEF_OP(VInsElement) {
auto Reg = GetVReg(Op->DestVector);
if (HostSupportsSVE256 && Is256Bit) {
// Broadcast our source value across a temporary, then combine
// with the destination.
//
// We don't need to perform the dup if we're just merging a 128-bit vector into
// into an equivalent position since we have a predicate set up already.
if (!(ElementSize == IR::OpSize::i128Bit && SrcIdx == DestIdx)) {
dup(SubRegSize, VTMP2.Z(), SrcVector.Z(), SrcIdx);
}
// Broadcast our source value across a temporary,
// then combine with the destination.
dup(SubRegSize, VTMP2.Z(), SrcVector.Z(), SrcIdx);
// We don't need to move the data unnecessarily if
// DestVector just so happens to also be the IR op
@@ -3082,12 +2995,10 @@ DEF_OP(VInsElement) {
if (ElementSize == IR::OpSize::i128Bit) {
if (DestIdx == 0) {
const auto Source = SrcIdx == 0 ? SrcVector : VTMP2;
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), PRED_TMP_16B.Merging(), Source.Z());
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), PRED_TMP_16B.Merging(), VTMP2.Z());
} else {
const auto Source = SrcIdx == 1 ? SrcVector : VTMP2;
not_(Predicate, PRED_TMP_32B.Zeroing(), PRED_TMP_16B);
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Predicate.Merging(), Source.Z());
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Predicate.Merging(), VTMP2.Z());
}
} else {
const auto UpperBound = 16 >> FEXCore::ilog2(IR::OpSizeToSize(ElementSize));
@@ -3222,12 +3133,19 @@ DEF_OP(VUShrI) {
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
} else {
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
if (BitShift == 0) {
if (Dst != Vector) {
mov(Dst.Z(), Vector.Z());
}
} else {
lsr(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
// SVE LSR is destructive, so lets set up the destination if
// Vector doesn't already alias it.
if (Dst != Vector) {
movprfx(Dst.Z(), Vector.Z());
}
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), BitShift);
}
} else {
if (BitShift == 0) {
@@ -3241,6 +3159,48 @@ DEF_OP(VUShrI) {
}
}
DEF_OP(VUShraI) {
const auto Op = IROp->C<IR::IROp_VUShraI>();
const auto OpSize = IROp->Size;
const auto BitShift = Op->BitShift;
const auto SubRegSize = ConvertSubRegSize8(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto DestVector = GetVReg(Op->DestVector);
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
if (Dst == DestVector) {
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
} else {
if (Dst != Vector) {
mov(Dst.Z(), DestVector.Z());
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
} else {
mov(VTMP1.Z(), DestVector.Z());
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
mov(Dst.Z(), VTMP1.Z());
}
}
} else {
if (Dst == DestVector) {
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
} else {
if (Dst != Vector) {
mov(Dst.Q(), DestVector.Q());
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
} else {
mov(VTMP1.Q(), DestVector.Q());
usra(SubRegSize, VTMP1.Q(), Vector.Q(), BitShift);
mov(Dst.Q(), VTMP1.Q());
}
}
}
}
DEF_OP(VSShrI) {
const auto Op = IROp->C<IR::IROp_VSShrI>();
const auto OpSize = IROp->Size;
@@ -3256,12 +3216,19 @@ DEF_OP(VSShrI) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
if (Shift == 0) {
if (Dst != Vector) {
mov(Dst.Z(), Vector.Z());
}
} else {
asr(SubRegSize, Dst.Z(), Vector.Z(), Shift);
// SVE ASR is destructive, so lets set up the destination if
// Vector doesn't already alias it.
if (Dst != Vector) {
movprfx(Dst.Z(), Vector.Z());
}
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), Shift);
}
} else {
if (Shift == 0) {
@@ -3291,12 +3258,19 @@ DEF_OP(VShlI) {
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
} else {
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
if (BitShift == 0) {
if (Dst != Vector) {
mov(Dst.Z(), Vector.Z());
}
} else {
lsl(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
// SVE LSL is destructive, so lets set up the destination if
// Vector doesn't already alias it.
if (Dst != Vector) {
movprfx(Dst.Z(), Vector.Z());
}
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), BitShift);
}
} else {
if (BitShift == 0) {
@@ -3323,13 +3297,8 @@ DEF_OP(VUShrNI) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
if (BitShift == 0) {
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
uzp1(SubRegSize, Dst.Z(), Dst.Z(), VTMP1.Z());
} else {
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
}
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
} else {
if (BitShift == 0) {
xtn(SubRegSize, Dst.D(), Vector.D());
@@ -3377,55 +3346,6 @@ DEF_OP(VUShrNI2) {
}
}
DEF_OP(VRSHRN) {
const auto Op = IROp->C<IR::IROp_VRSHRN>();
const auto OpSize = IROp->Size;
const auto BitShift = Op->BitShift;
const auto SubRegSize = ConvertSubRegSize4(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
rshrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
} else {
rshrn(SubRegSize, Dst.D(), Vector.D(), BitShift);
}
}
DEF_OP(VRSHRNPair) {
const auto Op = IROp->C<IR::IROp_VRSHRNPair>();
const auto OpSize = IROp->Size;
const auto BitShift = Op->BitShift;
const auto SubRegSize = ConvertSubRegSize4(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto VectorLower = GetVReg(Op->VectorLower);
auto VectorUpper = GetVReg(Op->VectorUpper);
if (HostSupportsSVE256 && Is256Bit) {
rshrnb(SubRegSize, VTMP1.Z(), VectorLower.Z(), BitShift);
rshrnb(SubRegSize, VTMP2.Z(), VectorUpper.Z(), BitShift);
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP2.Z());
} else {
if (Dst == VectorUpper) {
// RSHRN writes the lower half and would destroy the upper input.
mov(VTMP1.Q(), VectorUpper.Q());
VectorUpper = VTMP1;
}
rshrn(SubRegSize, Dst.D(), VectorLower.D(), BitShift);
rshrn2(SubRegSize, Dst.Q(), VectorUpper.Q(), BitShift);
}
}
DEF_OP(VSXTL) {
const auto Op = IROp->C<IR::IROp_VSXTL>();
const auto OpSize = IROp->Size;
@@ -3631,13 +3551,9 @@ DEF_OP(VSQXTN2) {
mov(Dst.Q(), VectorLower.Q());
ins(ARMEmitter::SubRegSize::i32Bit, Dst, 1, VTMP2, 0);
} else {
if (Dst == VectorLower) {
sqxtn2(SubRegSize, VectorLower, VectorUpper);
} else {
mov(VTMP1.Q(), VectorLower.Q());
sqxtn2(SubRegSize, VTMP1, VectorUpper);
mov(Dst.Q(), VTMP1.Q());
}
mov(VTMP1.Q(), VectorLower.Q());
sqxtn2(SubRegSize, VTMP1, VectorUpper);
mov(Dst.Q(), VTMP1.Q());
}
}
}
@@ -3875,171 +3791,6 @@ DEF_OP(VMul) {
}
}
DEF_OP(VUSDot) {
///< Dest = Acc + dot(Vector1 (unsigned 8-bit), Vector2 (signed 8-bit))
// Matches:
// - SVE - USDOT
// - ASIMD - USDOT
const auto Op = IROp->C<IR::IROp_VUSDot>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Acc = GetVReg(Op->Acc);
const auto Vector1 = GetVReg(Op->Vector1);
const auto Vector2 = GetVReg(Op->Vector2);
// USDOT accumulates in to its destination register,
// so we need to emit a move if Acc != Dst
ARMEmitter::VRegister DestTmp = Dst;
if (Dst != Acc) {
if (Dst != Vector1 && Dst != Vector2) {
DestTmp = Dst;
} else {
DestTmp = VTMP1;
}
}
if (HostSupportsSVE256 && Is256Bit) {
if (Dst != Acc) {
mov(DestTmp.Z(), Acc.Z());
}
usdot(DestTmp.Z(), Vector1.Z(), Vector2.Z());
if (Dst != DestTmp) {
mov(Dst.Z(), DestTmp.Z());
}
} else {
if (Dst != Acc) {
mov(DestTmp.Q(), Acc.Q());
}
usdot(DestTmp.Q(), Vector1.Q(), Vector2.Q());
if (Dst != DestTmp) {
mov(Dst.Q(), DestTmp.Q());
}
}
}
DEF_OP(VSDot) {
///< Dest = Acc + dot(Vector1 (signed 8-bit), Vector2 (signed 8-bit))
// Matches:
// - SVE - SDOT
// - ASIMD - SDOT
const auto Op = IROp->C<IR::IROp_VSDot>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Acc = GetVReg(Op->Acc);
const auto Vector1 = GetVReg(Op->Vector1);
const auto Vector2 = GetVReg(Op->Vector2);
// SDOT accumulates in to its destination register,
// so we need to emit a move if Acc != Dst
ARMEmitter::VRegister DestTmp = Dst;
if (Dst != Acc) {
if (Dst != Vector1 && Dst != Vector2) {
DestTmp = Dst;
} else {
DestTmp = VTMP1;
}
}
if (HostSupportsSVE256 && Is256Bit) {
if (Dst != Acc) {
mov(DestTmp.Z(), Acc.Z());
}
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Z(), Vector1.Z(), Vector2.Z());
if (Dst != DestTmp) {
mov(Dst.Z(), DestTmp.Z());
}
} else {
if (Dst != Acc) {
mov(DestTmp.Q(), Acc.Q());
}
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Q(), Vector1.Q(), Vector2.Q());
if (Dst != DestTmp) {
mov(Dst.Q(), DestTmp.Q());
}
}
}
DEF_OP(VSAddLP) {
const auto Op = IROp->C<IR::IROp_VSAddLP>();
const auto OpSize = IROp->Size;
const auto SubRegSize = ConvertSubRegSize248(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
// SVE only has the accumulating form, so accumulate in to a zeroed register.
// Zero a temporary instead if Dst aliases the source.
const auto DestTmp = Dst == Vector ? VTMP1 : Dst;
dup_imm(SubRegSize, DestTmp.Z(), 0);
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
if (Dst != DestTmp) {
mov(Dst.Z(), DestTmp.Z());
}
} else {
saddlp(SubRegSize, Dst.Q(), Vector.Q());
}
}
DEF_OP(VSAdALP) {
const auto Op = IROp->C<IR::IROp_VSAdALP>();
const auto OpSize = IROp->Size;
const auto SubRegSize = ConvertSubRegSize248(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Acc = GetVReg(Op->Acc);
const auto Vector = GetVReg(Op->Vector);
// SADALP accumulates in to its destination register,
// so we need to emit a move if Acc != Dst
ARMEmitter::VRegister DestTmp = Dst;
if (Dst != Acc) {
DestTmp = Dst != Vector ? Dst : VTMP1;
}
if (HostSupportsSVE256 && Is256Bit) {
if (Dst != Acc) {
mov(DestTmp.Z(), Acc.Z());
}
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
if (Dst != DestTmp) {
mov(Dst.Z(), DestTmp.Z());
}
} else {
if (Dst == Vector) {
// ASIMD has the non-accumulating form, which is cheaper than shuffling through a temporary.
saddlp(SubRegSize, VTMP1.Q(), Vector.Q());
add(SubRegSize, Dst.Q(), Acc.Q(), VTMP1.Q());
return;
}
if (Dst != Acc) {
mov(Dst.Q(), Acc.Q());
}
sadalp(SubRegSize, Dst.Q(), Vector.Q());
}
}
DEF_OP(VUMull) {
const auto Op = IROp->C<IR::IROp_VUMull>();
const auto OpSize = IROp->Size;
@@ -4658,9 +4409,13 @@ DEF_OP(VFMLS) {
if (Is128Bit) {
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
fmla(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
} else {
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
}
if (Is128Bit) {
fmla(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
} else {
fmla(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
}
@@ -4680,7 +4435,7 @@ DEF_OP(VFNMLA) {
// - SVE - FMLS
// - ASIMD - FMLS
// - Scalar - FMSUB
const auto Op = IROp->C<IR::IROp_VFNMLA>();
const auto Op = IROp->C<IR::IROp_VFMLA>();
const auto OpSize = IROp->Size;
const auto SubRegSize = ConvertSubRegSize248(IROp);
@@ -4748,7 +4503,7 @@ DEF_OP(VFNMLS) {
// - ASIMD - FMLS (With Negated addend)
// - Scalar - FNMADD
const auto Op = IROp->C<IR::IROp_VFNMLS>();
const auto Op = IROp->C<IR::IROp_VFMLS>();
const auto OpSize = IROp->Size;
const auto SubRegSize = ConvertSubRegSize248(IROp);
@@ -4814,9 +4569,13 @@ DEF_OP(VFNMLS) {
if (Is128Bit) {
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
fmls(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
} else {
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
}
if (Is128Bit) {
fmls(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
} else {
fmls(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
}
@@ -4830,106 +4589,6 @@ DEF_OP(VFNMLS) {
}
}
DEF_OP(VBlendImm) {
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "Host must support SVE to use {}", __func__);
auto Op = IROp->C<IR::IROp_VBlendImm>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
const auto SubRegSize = ConvertSubRegSize8(IROp);
const auto ElementSize = IROp->ElementSize;
const auto Selector = Op->Selector;
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
const auto Dst = GetVReg(Node);
const auto LHS = GetVReg(Op->LHS);
const auto RHS = GetVReg(Op->RHS);
const auto DstIsNonAliasing = Dst != LHS && Dst != RHS;
// Silly case where two blending sources are the same.
if (LHS == RHS) {
if (DstIsNonAliasing) {
mov(SubRegSize, Dst.Z(), GoverningPredicate.Merging(), LHS.Z());
}
return;
}
// We'll need to expand our selector to match its predicate equivalent.
// The lowest bit of each predicate element being set to 1 signifies
// that it's enabled.
const auto MakePredicateMask = [ElementSize, Is256Bit, OpSize](uint16_t Imm) {
if (ElementSize == IR::OpSize::i8Bit) {
// Since we use a u16 selector, we have enough bits for every byte in a
// 128-bit lane, so we don't need to do anything here except replicate the
// bits in the event of 256-bit.
return Is256Bit ? uint32_t(Imm) << 16 | Imm : Imm;
}
uint32_t Mask = 0;
const auto DataSize = IR::OpSizeToSize(ElementSize);
const auto NumElements = IR::NumElements(OpSize, ElementSize);
for (uint32_t i = 0; i < NumElements; i++) {
if (((Imm >> i) & 1) != 0) {
Mask |= 1U << (DataSize * i);
}
}
return Mask;
};
// Our predicate that we'll be firing our constructed bitmask into.
constexpr auto Predicate = ARMEmitter::PReg::p0.Merging();
// TODO: We can completely eliminate this via PMOV in SVE2.1
ARMEmitter::ForwardLabel AfterLabel;
ARMEmitter::BackwardLabel ConstantLabel;
(void)b(&AfterLabel);
(void)Bind(&ConstantLabel);
const auto PredicateMask = MakePredicateMask(Selector);
if (Dst == RHS) {
dc32(~PredicateMask);
} else {
dc32(PredicateMask);
}
(void)Bind(&AfterLabel);
(void)adr(TMP1, &ConstantLabel);
ldr(Predicate, TMP1);
if (Dst == LHS) {
mov(SubRegSize, LHS.Z(), Predicate, RHS.Z());
} else if (Dst == RHS) {
mov(SubRegSize, RHS.Z(), Predicate, LHS.Z());
} else {
mov(SubRegSize, Dst.Z(), GoverningPredicate.Merging(), LHS.Z());
mov(SubRegSize, Dst.Z(), Predicate, RHS.Z());
}
}
DEF_OP(VXar) {
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "Host must support SVE to use {}", __func__);
auto Op = IROp->C<IR::IROp_VXar>();
const auto SubRegSize = ConvertSubRegSize8(IROp);
const auto ElementSizeBits = IR::OpSizeAsBits(IROp->ElementSize);
const auto Dst = GetVReg(Node);
const auto LHS = GetVReg(Op->LHS);
const auto RHS = GetVReg(Op->RHS);
const auto Rotate = Op->Rotate;
LOGMAN_THROW_A_FMT(Rotate >= 1 && Rotate <= ElementSizeBits, "Rotate immediate must be within [1, {}]", ElementSizeBits);
if (Dst == LHS) {
xar(SubRegSize, Dst.Z(), RHS.Z(), Rotate);
} else if (Dst == RHS) {
movprfx(VTMP1.Z(), LHS.Z());
xar(SubRegSize, VTMP1.Z(), RHS.Z(), Rotate);
mov(Dst.Z(), VTMP1.Z());
} else {
movprfx(Dst.Z(), LHS.Z());
xar(SubRegSize, Dst.Z(), RHS.Z(), Rotate);
}
}
DEF_OP(VFCopySign) {
auto Op = IROp->C<IR::IROp_VFCopySign>();
const auto OpSize = IROp->Size;
@@ -4953,36 +4612,6 @@ DEF_OP(VFCopySign) {
}
}
DEF_OP(F64FPREM) {
const auto Op = IROp->C<IR::IROp_F64FPREM>();
const auto Dst = GetVReg(Node);
const auto Src1 = GetVReg(Op->Src1);
const auto Src2 = GetVReg(Op->Src2);
fmov(VTMP1.D(), Src1.D());
fmov(VTMP2.D(), Src2.D());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FPREMHandler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
fmov(Dst.D(), VTMP1.D());
}
DEF_OP(F64FPREM1) {
const auto Op = IROp->C<IR::IROp_F64FPREM1>();
const auto Dst = GetVReg(Node);
const auto Src1 = GetVReg(Op->Src1);
const auto Src2 = GetVReg(Op->Src2);
fmov(VTMP1.D(), Src1.D());
fmov(VTMP2.D(), Src2.D());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FPREM1Handler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
fmov(Dst.D(), VTMP1.D());
}
DEF_OP(F64SIN) {
const auto Op = IROp->C<IR::IROp_F64SIN>();
const auto Src = GetVReg(Op->Src);
@@ -5054,22 +4683,6 @@ DEF_OP(F64FYL2X) {
fmov(Dst.D(), VTMP1.D());
}
// Src=x(ST0), Src2=y(ST1). Marshal into VTMP1/VTMP2 and dispatch the shared handler.
DEF_OP(F64FYL2XP1) {
const auto Op = IROp->C<IR::IROp_F64FYL2XP1>();
const auto Src = GetVReg(Op->Src);
const auto Src2 = GetVReg(Op->Src2);
const auto Dst = GetVReg(Node);
fmov(VTMP1.D(), Src.D());
fmov(VTMP2.D(), Src2.D());
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FYL2XP1Handler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP1);
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
fmov(Dst.D(), VTMP1.D());
}
DEF_OP(F64SCALE) {
const auto Op = IROp->C<IR::IROp_F64SCALE>();
const auto Src1 = GetVReg(Op->Src1);
@@ -87,11 +87,8 @@ void LookupCache::ClearL2Cache(const FEXCore::LookupCacheBaseLockToken& lk) {
}
void LookupCache::ClearThreadLocalCaches(const LookupCacheWriteLockToken&) {
// TODO: Preserve code cache entries?
// Clear L1 and L2 by clearing the full cache.
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
// TODO: Rename this member to avoid confusion with code caching
CachedCodePages.clear();
}
+9 -22
View File
@@ -13,7 +13,6 @@
#include <FEXCore/fextl/memory_resource.h>
#include <cstdint>
#include <span>
#include <stddef.h>
#include <utility>
#include <mutex>
@@ -94,15 +93,13 @@ struct GuestToHostMap {
GuestToHostMap();
// Adds to Guest -> Host code mapping
const BlockEntry& AddBlockMapping(uint64_t Address, std::span<const uint64_t> CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
const BlockEntry& AddBlockMapping(uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
// This may replace an existing mapping
// NOTE: Generally no previous entry should exist, however there is one exception:
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
// may already contain the block address. Since is comparatively rare, we'll just leak
// one of the two blocks in this case.
return BlockList
.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, fextl::vector<uint64_t>(CodePages.begin(), CodePages.end())})
.first->second;
return BlockList.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, CodePages}).first->second;
}
const BlockEntry* FindBlock(uint64_t Address, const LookupCacheReadLockToken&) {
@@ -223,7 +220,7 @@ public:
}
if (HostPtr && DynamicL1Cache()) {
UpdateDynamicL1Stats(Thread, Address, HostPtr);
UpdateDynamicL1Stats(Thread);
}
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedCacheMissCount, 1);
@@ -231,7 +228,7 @@ public:
return HostPtr;
}
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddress, uint64_t HostCode) {
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread) {
// If host pointer was found in L2 or L3, then add it to the counter.
// Keeping track not L1 misses, but specifically L2/L3 hits.
++L2L3CacheHits;
@@ -245,18 +242,12 @@ public:
if (AveragePerSecond >= DynamicL1CacheIncreaseCountHeuristic()) {
if (CurrentL1Entries < MAX_L1_ENTRIES) {
// Entries whose address has the new mask bit set would be unreachable by InvalidateCache
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(L1Pointer), CurrentL1Entries * sizeof(LookupCacheEntry), false);
CurrentL1Entries <<= 1;
L1PointerMask = CurrentL1Entries - 1;
// Update the thread's L1 pointer mask to increase how much cache it uses.
// Since we're in C-code, this is safe to update here.
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
// If L1 was just shrunk, then we just removed our cached entry. Add it back.
AddL1Entry(GuestAddress, HostCode);
}
} else if (AveragePerSecond < DynamicL1CacheDecreaseCountHeuristic()) {
if (CurrentL1Entries > MIN_L1_ENTRIES) {
@@ -284,7 +275,7 @@ public:
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
// Returns true if new pages are marked as containing code
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, auto& Addresses, uint64_t Start, uint64_t Length) {
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
auto lk = Shared->AcquireWriteLock();
@@ -294,7 +285,7 @@ public:
}
// Adds to Guest -> Host code mapping
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, std::span<const uint64_t> CodePages, void* HostCode) {
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode) {
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
auto lk = Shared->AcquireWriteLock();
@@ -389,19 +380,15 @@ public:
}
private:
void AddL1Entry(uint64_t GuestAddress, uint64_t HostCode) {
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[GuestAddress & L1PointerMask];
L1Entry.GuestCode = GuestAddress;
L1Entry.HostCode = HostCode;
}
void CacheBlockMapping(uint64_t Address, const GuestToHostMap::BlockEntry& Entry, bool L1Only, const LookupCacheBaseLockToken& lk) {
for (const auto& CodePage : Entry.CodePages) {
CachedCodePages[CodePage >> 12].insert(Address);
}
// Do L1
AddL1Entry(Address, Entry.HostCode);
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
L1Entry.GuestCode = Address;
L1Entry.HostCode = Entry.HostCode;
if (!DisableL2Cache() && !L1Only) {
// Do ful map
File diff suppressed because it is too large. Load diff
+159 -140
View File
@@ -202,10 +202,9 @@ public:
FlushRegisterCache();
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, InvalidNode, InvalidNode);
}
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock,
uint64_t PatchSiteAddress = 0, uint64_t PatchSiteSize = 0) {
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
FlushRegisterCache();
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, CallReturnAddress, CallReturnBlock, PatchSiteAddress, PatchSiteSize);
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, CallReturnAddress, CallReturnBlock);
}
IRPair<IROp_Break> Break(BreakDefinition Reason) {
FlushRegisterCache();
@@ -304,7 +303,7 @@ public:
StartNewBlock();
}
OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx);
// Should only be called at the start of IR Emission.
void ResetWorkingList();
@@ -361,10 +360,8 @@ public:
void MOVGPRNTOp(OpcodeArgs);
void MOVVectorAlignedOp(OpcodeArgs);
void MOVVectorUnalignedOp(OpcodeArgs);
void MOVVectorUnalignedNoNopOp(OpcodeArgs);
void MOVVectorNTOp(OpcodeArgs, bool IsAVX);
void ALURAXOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp);
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx, bool DestRAX);
void MOVVectorNTOp(OpcodeArgs);
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
void LSLOp(OpcodeArgs);
void INTOp(OpcodeArgs);
void SyscallOp(OpcodeArgs, bool IsSyscallInst);
@@ -375,8 +372,8 @@ public:
void IRETOp(OpcodeArgs);
void CallbackReturnOp(OpcodeArgs);
void SecondaryALUOp(OpcodeArgs);
void ADCOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
void SBBOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
void ADCOp(OpcodeArgs, uint32_t SrcIndex);
void SBBOp(OpcodeArgs, uint32_t SrcIndex);
void SALCOp(OpcodeArgs);
void PUSHOp(OpcodeArgs);
void PUSHREGOp(OpcodeArgs);
@@ -396,18 +393,16 @@ public:
void JUMPFARIndirectOp(OpcodeArgs);
void CALLFARIndirectOp(OpcodeArgs);
void RETFARIndirectOp(OpcodeArgs);
void TESTOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
void TESTOp(OpcodeArgs, uint32_t SrcIndex);
void ARPLOp(OpcodeArgs);
void MOVSXDOp(OpcodeArgs);
void MOVSXOp(OpcodeArgs);
void MOVZXOp(OpcodeArgs);
void CMPOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
void CMPOp(OpcodeArgs, uint32_t SrcIndex);
void SETccOp(OpcodeArgs);
void CQOOp(OpcodeArgs);
void CDQOp(OpcodeArgs);
std::optional<Ref> XCHGOpImpl(OpcodeArgs, Ref Src);
void XCHGOp(OpcodeArgs);
void XCHGRAXOp(OpcodeArgs);
void SAHFOp(OpcodeArgs);
void LAHFOp(OpcodeArgs);
void MOVSegOp(OpcodeArgs, bool ToSeg);
@@ -429,11 +424,11 @@ public:
void RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool Is1Bit);
void RCROp1Bit(OpcodeArgs);
void RCROp8x1Bit(OpcodeArgs);
void RCROp(OpcodeArgs, bool UseRCX);
void RCRSmallerOp(OpcodeArgs, bool UseRCX);
void RCROp(OpcodeArgs);
void RCRSmallerOp(OpcodeArgs);
void RCLOp1Bit(OpcodeArgs);
void RCLOp(OpcodeArgs, bool UseRCX);
void RCLSmallerOp(OpcodeArgs, bool UseRCX);
void RCLOp(OpcodeArgs);
void RCLSmallerOp(OpcodeArgs);
void BTOp(OpcodeArgs, uint32_t SrcIndex, enum BTAction Action);
@@ -475,7 +470,8 @@ public:
void AAMOp(OpcodeArgs);
void AADOp(OpcodeArgs);
void XLATOp(OpcodeArgs);
void RDRANDOp(OpcodeArgs, bool Reseed);
template<bool Reseed>
void RDRANDOp(OpcodeArgs);
enum class Segment {
FS,
@@ -505,7 +501,8 @@ public:
void VectorALUROp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
void VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
void RSqrt3DNowOp(OpcodeArgs, bool Duplicate);
void VectorUnaryDuplicateOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
void VectorUnaryDuplicateOp(OpcodeArgs);
void MOVQOp(OpcodeArgs, VectorOpType VectorType);
void MOVQMMXOp(OpcodeArgs);
@@ -527,24 +524,36 @@ public:
void PSLLDQ(OpcodeArgs);
void PSRAIOp(OpcodeArgs, IR::OpSize ElementSize);
void MOVDDUPOp(OpcodeArgs);
void CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode);
void Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen, bool IsAVX);
template<IR::OpSize DstElementSize>
void CVTGPR_To_FPR(OpcodeArgs);
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
void CVTFPR_To_GPR(OpcodeArgs);
template<IR::OpSize SrcElementSize, bool Widen>
void Vector_CVT_Int_To_Float(OpcodeArgs);
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
void Scalar_CVT_Float_To_Float(OpcodeArgs);
void Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize, bool IsAVX);
void Vector_CVT_Float_To_Int(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode, bool IsAVX);
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
void Vector_CVT_Float_To_Int(OpcodeArgs);
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode);
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
void MASKMOVOp(OpcodeArgs);
void MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType);
void TZCNT(OpcodeArgs);
void LZCNT(OpcodeArgs);
void VFCMPOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void VFCMPOp(OpcodeArgs);
void SHUFOp(OpcodeArgs, IR::OpSize ElementSize);
void PINSROp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void PINSROp(OpcodeArgs);
void InsertPSOp(OpcodeArgs);
void PExtrOp(OpcodeArgs, IR::OpSize ElementSize);
void PSIGN(OpcodeArgs, IR::OpSize ElementSize);
void VPSIGN(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void PSIGN(OpcodeArgs);
template<IR::OpSize ElementSize>
void VPSIGN(OpcodeArgs);
// BMI1 Ops
void ANDNBMIOp(OpcodeArgs);
@@ -567,32 +576,53 @@ public:
// AVX Ops
void AVXVectorXOROp(OpcodeArgs);
void AVXVectorRound(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVXVectorRound(OpcodeArgs);
void VectorScalarInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
void AVXVectorScalarInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
void AVXScalar_CVT_Float_To_Float(OpcodeArgs);
void VectorScalarUnaryInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
void AVXVectorScalarUnaryInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
void VectorScalarInsertALUOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
void AVXVectorScalarInsertALUOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
void VectorScalarUnaryInsertALUOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
void AVXVectorScalarUnaryInsertALUOp(OpcodeArgs);
void InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
void InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstElementSize);
void AVXInsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstElementSize);
template<IR::OpSize DstElementSize>
void InsertCVTGPR_To_FPR(OpcodeArgs);
template<IR::OpSize DstElementSize>
void AVXInsertCVTGPR_To_FPR(OpcodeArgs);
void InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize);
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize);
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
void InsertScalar_CVT_Float_To_Float(OpcodeArgs);
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
RoundMode TranslateRoundType(uint8_t Mode);
void InsertScalarRound(OpcodeArgs, IR::OpSize ElementSize);
void AVXInsertScalarRound(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void InsertScalarRound(OpcodeArgs);
template<IR::OpSize ElementSize>
void AVXInsertScalarRound(OpcodeArgs);
void InsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize);
void AVXInsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void InsertScalarFCMPOp(OpcodeArgs);
template<IR::OpSize ElementSize>
void AVXInsertScalarFCMPOp(OpcodeArgs);
void AVXVFCMPOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize DstElementSize>
void AVXCVTGPR_To_FPR(OpcodeArgs);
void VADDSUBPOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVXVFCMPOp(OpcodeArgs);
template<IR::OpSize ElementSize>
void VADDSUBPOp(OpcodeArgs);
void VAESDecOp(OpcodeArgs);
void VAESDecLastOp(OpcodeArgs);
@@ -601,31 +631,34 @@ public:
void VANDNOp(OpcodeArgs);
Ref VBLENDOpImpl(IR::OpSize VecSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint64_t Selector);
Ref VBLENDOpImpl(IR::OpSize VecSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref ZeroRegister, uint64_t Selector);
void VBLENDPDOp(OpcodeArgs);
void VPBLENDDOp(OpcodeArgs);
void VPBLENDWOp(OpcodeArgs);
void VBROADCASTOp(OpcodeArgs, IR::OpSize ElementSize);
void VDPPOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void VDPPOp(OpcodeArgs);
void VEXTRACT128Op(OpcodeArgs);
void VHADDPOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
template<IROps IROp, IR::OpSize ElementSize>
void VHADDPOp(OpcodeArgs);
void VHSUBPOp(OpcodeArgs, IR::OpSize ElementSize);
void VINSERTOp(OpcodeArgs);
void VINSERTPSOp(OpcodeArgs);
void VMASKMOVOp(OpcodeArgs, IR::OpSize ElementSize, bool IsStore);
template<IR::OpSize ElementSize, bool IsStore>
void VMASKMOVOp(OpcodeArgs);
void VMOVHPOp(OpcodeArgs);
void VMOVLPOp(OpcodeArgs);
void VMOVDDUPOp(OpcodeArgs);
void VMOVSHDUPOp(OpcodeArgs, bool IsAVX);
void VMOVSLDUPOp(OpcodeArgs, bool IsAVX);
void VMOVSHDUPOp(OpcodeArgs);
void VMOVSLDUPOp(OpcodeArgs);
void VMOVSDOp(OpcodeArgs);
void VMOVSSOp(OpcodeArgs);
@@ -636,14 +669,15 @@ public:
void VMPSADBWOp(OpcodeArgs);
void VPACKSSOp(OpcodeArgs, IR::OpSize ElementSize);
void VPACKUSOp(OpcodeArgs, IR::OpSize ElementSize);
void VPALIGNROp(OpcodeArgs);
void VPCMPESTRIOp(OpcodeArgs, bool IsAVX);
void VPCMPESTRMOp(OpcodeArgs, bool IsAVX);
void VPCMPISTRIOp(OpcodeArgs, bool IsAVX);
void VPCMPISTRMOp(OpcodeArgs, bool IsAVX);
void VPCMPESTRIOp(OpcodeArgs);
void VPCMPESTRMOp(OpcodeArgs);
void VPCMPISTRIOp(OpcodeArgs);
void VPCMPISTRMOp(OpcodeArgs);
void VCVTPH2PSOp(OpcodeArgs);
void VCVTPS2PHOp(OpcodeArgs);
@@ -656,30 +690,36 @@ public:
void VPERMILImmOp(OpcodeArgs, IR::OpSize ElementSize);
Ref VPERMILRegOpImpl(OpSize DstSize, IR::OpSize ElementSize, Ref Src, Ref Indices);
void VPERMILRegOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void VPERMILRegOp(OpcodeArgs);
void VPHADDSWOp(OpcodeArgs);
void VPHSUBOp(OpcodeArgs, IR::OpSize ElementSize);
void VPHSUBSWOp(OpcodeArgs);
void VPINSRBWOp(OpcodeArgs, IR::OpSize ElementSize);
void VPINSRBOp(OpcodeArgs);
void VPINSRDQOp(OpcodeArgs);
void VPINSRWOp(OpcodeArgs);
void VPMADDUBSWOp(OpcodeArgs);
void VPMADDWDOp(OpcodeArgs);
void VPDPBUSDOp(OpcodeArgs, bool Saturating);
void VPDPWSSDOp(OpcodeArgs, bool Saturating);
void VPMASKMOVOp(OpcodeArgs, bool IsStore);
template<bool IsStore>
void VPMASKMOVOp(OpcodeArgs);
void VPMULHRSWOp(OpcodeArgs);
void VPMULHWOp(OpcodeArgs, bool Signed);
void VPMULLOp(OpcodeArgs, IR::OpSize ElementSize, bool Signed);
template<bool Signed>
void VPMULHWOp(OpcodeArgs);
template<IR::OpSize ElementSize, bool Signed>
void VPMULLOp(OpcodeArgs);
void VPSADBWOp(OpcodeArgs);
void VPSHUFBOp(OpcodeArgs);
void VPSHUFWOp(OpcodeArgs, IR::OpSize ElementSize, bool Low);
void VPSLLOp(OpcodeArgs, IR::OpSize ElementSize);
@@ -688,6 +728,7 @@ public:
void VPSLLVOp(OpcodeArgs);
void VPSRAOp(OpcodeArgs, IR::OpSize ElementSize);
void VPSRAIOp(OpcodeArgs, IR::OpSize ElementSize);
void VPSRAVDOp(OpcodeArgs);
@@ -695,14 +736,17 @@ public:
void VPSRLDOp(OpcodeArgs, IR::OpSize ElementSize);
void VPSRLDQOp(OpcodeArgs);
void VPSRLIOp(OpcodeArgs, IR::OpSize ElementSize);
void VPUNPCKHOp(OpcodeArgs, IR::OpSize ElementSize);
void VPUNPCKLOp(OpcodeArgs, IR::OpSize ElementSize);
void VPSRLIOp(OpcodeArgs, IR::OpSize ElementSize);
void VSHUFOp(OpcodeArgs, IR::OpSize ElementSize);
void VTESTPOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void VTESTPOp(OpcodeArgs);
void VZEROOp(OpcodeArgs);
@@ -746,7 +790,7 @@ public:
void X87FLDCW(OpcodeArgs);
void X87FNSAVE(OpcodeArgs);
void X87FNSTENV(OpcodeArgs);
void X87FNSTSW(OpcodeArgs, bool DestRAX);
void X87FNSTSW(OpcodeArgs);
void X87FRSTOR(OpcodeArgs);
void X87FSTCW(OpcodeArgs);
void X87FXAM(OpcodeArgs);
@@ -786,39 +830,50 @@ public:
void XSaveOp(OpcodeArgs);
void PAlignrOp(OpcodeArgs);
void UCOMISxOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void UCOMISxOp(OpcodeArgs);
void LDMXCSR(OpcodeArgs);
void STMXCSR(OpcodeArgs);
void PACKUSOp(OpcodeArgs, IR::OpSize ElementSize);
void PACKSSOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void PACKUSOp(OpcodeArgs);
void PMULLOp(OpcodeArgs, IR::OpSize ElementSize, bool Signed);
template<IR::OpSize ElementSize>
void PACKSSOp(OpcodeArgs);
void MOVQ2DQ(OpcodeArgs, bool ToXMM);
template<IR::OpSize ElementSize, bool Signed>
void PMULLOp(OpcodeArgs);
void ADDSUBPOp(OpcodeArgs, IR::OpSize ElementSize);
template<bool ToXMM>
void MOVQ2DQ(OpcodeArgs);
template<IR::OpSize ElementSize>
void ADDSUBPOp(OpcodeArgs);
void PFNACCOp(OpcodeArgs);
void PFPNACCOp(OpcodeArgs);
void PSWAPDOp(OpcodeArgs);
void VPFCMPOp(OpcodeArgs, uint8_t CompType);
template<uint8_t CompType>
void VPFCMPOp(OpcodeArgs);
void PI2FWOp(OpcodeArgs);
void PF2IWOp(OpcodeArgs);
void PF2IDOp(OpcodeArgs);
void PMULHRWOp(OpcodeArgs);
void PMADDWD(OpcodeArgs);
void PMADDUBSW(OpcodeArgs);
void PMULHW(OpcodeArgs, bool Signed);
template<bool Signed>
void PMULHW(OpcodeArgs);
void PMULHRSW(OpcodeArgs);
void MOVBEOp(OpcodeArgs);
void HSUBP(OpcodeArgs, IR::OpSize ElementSize);
void PHSUB(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void HSUBP(OpcodeArgs);
template<IR::OpSize ElementSize>
void PHSUB(OpcodeArgs);
void PHADDS(OpcodeArgs);
void PHSUBS(OpcodeArgs);
@@ -847,12 +902,12 @@ public:
void SHA256MSG2Op(OpcodeArgs);
void SHA256RNDS2Op(OpcodeArgs);
void AESImcOp(OpcodeArgs, bool IsAVX);
void AESImcOp(OpcodeArgs);
void AESEncOp(OpcodeArgs);
void AESEncLastOp(OpcodeArgs);
void AESDecOp(OpcodeArgs);
void AESDecLastOp(OpcodeArgs);
void AESKeyGenAssist(OpcodeArgs, bool IsAVX);
void AESKeyGenAssist(OpcodeArgs);
void VFMAImpl(OpcodeArgs, IROps IROp, bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
void VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
@@ -865,24 +920,25 @@ public:
};
RefVSIB LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags);
void VPGATHER(OpcodeArgs, OpSize AddrElementSize);
template<OpSize AddrElementSize>
void VPGATHER(OpcodeArgs);
void AVXExtendVectorElements(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DstElementSize, bool Signed);
void ExtendVectorElements(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DstElementSize, bool Signed);
template<IR::OpSize ElementSize, IR::OpSize DstElementSize, bool Signed>
void ExtendVectorElements(OpcodeArgs);
template<IR::OpSize ElementSize>
void VectorRound(OpcodeArgs);
void VectorRound(OpcodeArgs, IR::OpSize ElementSize);
Ref VectorBlend(OpSize Size, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t Selector);
Ref VectorBlendImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t Selector);
void VectorBlend(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void VectorBlend(OpcodeArgs);
void VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize);
void PTestOpImpl(OpSize Size, Ref Dest, Ref Src);
void PTestOp(OpcodeArgs);
void AVXPHMINPOSUWOp(OpcodeArgs);
void PHMINPOSUWOp(OpcodeArgs);
void DPPOp(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void DPPOp(OpcodeArgs);
void MPSADBWOp(OpcodeArgs);
void PCLMULQDQOp(OpcodeArgs);
@@ -1041,9 +1097,6 @@ public:
void AVX128_VPMADDUBSW(OpcodeArgs);
void AVX128_VPMADDWD(OpcodeArgs);
void AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper);
void AVX128_VPDPBUSD(OpcodeArgs, bool Saturating);
void AVX128_VPDPWSSD(OpcodeArgs, bool Saturating);
void AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize);
@@ -1323,7 +1376,6 @@ private:
};
FEXCore::Context::ContextImpl* CTX {};
FEXCore::Core::InternalThreadState* Thread;
constexpr static unsigned FullNZCVMask = (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC) | (1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC) |
(1U << FEXCore::X86State::RFLAG_SF_RAW_LOC) | (1U << FEXCore::X86State::RFLAG_OF_RAW_LOC);
@@ -1391,12 +1443,7 @@ private:
Ref PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
const X86Tables::DecodedOperand& Imm, bool IsAVX);
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask, bool IsAVX);
Ref PCMPXSTRXSaturateExplicitLength(IR::OpSize ElementSize, IR::OpSize LengthSize, Ref RawLength);
Ref PCMPXSTRXEqualAny(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements);
Ref PCMPXSTRXRanges(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements, bool IsSigned);
Ref PCMPXSTRXEqualEach(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements);
Ref PCMPXSTRXEqualOrdered(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1Length, Ref Src2Length, Ref Src1ValidElements, Ref Indices);
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
Ref PHADDSOpImpl(OpSize Size, Ref Src1, Ref Src2);
@@ -1413,10 +1460,6 @@ private:
Ref PMADDUBSWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2);
Ref VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
Ref VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
Ref PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2);
Ref PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2);
@@ -1438,7 +1481,7 @@ private:
Ref PSRLDOpImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src, Ref ShiftVec);
Ref SHUFOpImpl(IR::OpSize DstSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t Shuffle);
Ref SHUFOpImpl(OpcodeArgs, IR::OpSize DstSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t Shuffle);
void VMASKMOVOpImpl(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DataSize, bool IsStore, const X86Tables::DecodedOperand& MaskOp,
const X86Tables::DecodedOperand& DataOp);
@@ -1518,20 +1561,6 @@ private:
void StoreGPRRegister(uint32_t GPR, const Ref Src, IR::OpSize Size = OpSize::iInvalid, uint8_t Offset = 0);
void StoreXMMRegister(uint32_t XMM, const Ref Src);
// Matches semantics around GPR storing that matches `StoreResult_WithOpSize` behaviour.
void StoreGPRResultWithZExtSemantics(uint32_t GPR, Ref Src, IR::OpSize OpSize) {
const auto GPRSize = GetGPROpSize();
if (GPRSize == OpSize::i64Bit && OpSize == OpSize::i32Bit) {
// If the Source IR op is 64 bits, we need to zext the upper bits
// For all other sizes, the upper bits are guaranteed to already be zero
Src = GetOpSize(Src) == OpSize::i64Bit ? ARef(Src).Bfe(0, 32).Ref() : Src;
StoreGPRRegister(GPR, Src, GPRSize);
} else {
StoreGPRRegister(GPR, Src, std::min(GPRSize, OpSize));
}
}
Ref _GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, bool Inline) {
const auto GPRSize = GetGPROpSize();
const auto Offs = Op->PC + Op->InstSize + Offset - Entry;
@@ -1542,24 +1571,18 @@ private:
return _GetRelocatedPC(Op, Offset, false);
}
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
uint64_t PatchOffset = 0;
uint64_t PatchSize = 0;
if (Op->Src[0].IsLiteralPatchable() && Offset && Offset == (int64_t)Op->Src[0].Literal()) {
PatchOffset = Op->PC + Op->Src[0].Data.LiteralPatchable.FieldOffset;
PatchSize = Op->Src[0].Data.LiteralPatchable.Width;
}
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock, PatchOffset, PatchSize);
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */));
}
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
ExitRelocatedPC(Op, Offset, BranchHint::None, InvalidNode, InvalidNode);
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock);
}
[[nodiscard]]
static bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
// Literals are immediates as sources but memory addresses as destinations.
return !(Load && (Operand.IsLiteral() || Operand.IsLiteralRelocation() || Operand.IsLiteralPatchable())) && !Operand.IsGPR();
return !(Load && (Operand.IsLiteral() || Operand.IsLiteralRelocation())) && !Operand.IsGPR();
}
[[nodiscard]]
@@ -1568,7 +1591,6 @@ private:
}
AddressMode DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, MemoryAccessType AccessType, bool IsLoad);
uint64_t CalcAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, bool IsLoad);
Ref LoadSource(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
const LoadSourceOptions& Options = {});
@@ -2171,16 +2193,16 @@ private:
}
// Compares two floats and sets flags for a COMISS instruction
void Comiss(IR::OpSize ElementSize, Ref Src1, Ref Src2) {
void Comiss(IR::OpSize ElementSize, Ref Src1, Ref Src2, bool InvalidateAF = false) {
// First, set flags according to Arm FCMP.
HandleNZCVWrite();
_FCmp(ElementSize, Src1, Src2);
CFInverted = false;
ComissFlags();
ComissFlags(InvalidateAF);
}
// Sets flags for a COMISS instruction
void ComissFlags() {
void ComissFlags(bool InvalidateAF = false) {
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
// We need to set PF according to the unordered flag. We'd rather do this
@@ -2195,15 +2217,12 @@ private:
Ref V_inv = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC, true);
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(V_inv);
// Intel: OF, SF, and AF set to zero
// AMD: no mention of OF, SF and AF but actual hardware seems to always zero
//
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
// byte to zero will indeed zero AF as intended.
// OF and SF are zeroed:
// _AXFLAG always produces N=0 (SF), V=0 (OF)
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
if (!InvalidateAF) {
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
// byte to zero will indeed zero AF as intended.
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
}
// Convert NZCV from the Arm representation to an eXternal representation
// that's totally not a euphemism for x86, nuh-uh. But maps to exactly we
@@ -2415,7 +2434,7 @@ private:
void CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High);
void CalculateFlags_UMUL(Ref High);
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res);
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift, bool DoubleWide = false);
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
void CalculateFlags_ShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
void CalculateFlags_ShiftRightDoubleImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
void CalculateFlags_ShiftRightImmediateCommon(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
@@ -603,7 +603,7 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), OpSize::i128Bit,
[this](IR::OpSize, Ref Src1, Ref Src2) { return _VAndn(OpSize::i128Bit, Src2, Src1); });
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return _VAndn(OpSize::i128Bit, _ElementSize, Src2, Src1); });
}
void OpDispatchBuilder::AVX128_VPACKSS(OpcodeArgs, IR::OpSize ElementSize) {
@@ -630,7 +630,7 @@ void OpDispatchBuilder::AVX128_VPSIGN(OpcodeArgs, IR::OpSize ElementSize) {
}
void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize) {
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : ElementSize;
const auto SrcSize = Op->Src[0].IsGPR() ? GetGuestVectorLength() : ElementSize;
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, false);
@@ -865,14 +865,13 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
GPR = Mask4Byte(Src.Low);
}
} else if (ElementSize == OpSize::i32Bit) {
Ref Fused = _VUnZip2(OpSize::i128Bit, OpSize::i16Bit, Src.Low, Src.High);
Fused = _VUShrI(OpSize::i128Bit, OpSize::i16Bit, Fused, 15);
auto ConstantUSHL = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_INCREMENTAL_U16_INDEX);
Fused = _VUShl(OpSize::i128Bit, OpSize::i16Bit, Fused, ConstantUSHL, false);
Fused = _VAddV(OpSize::i128Bit, OpSize::i16Bit, Fused);
GPR = _VExtractToGPR(OpSize::i128Bit, OpSize::i16Bit, Fused, 0);
auto GPRLow = Mask4Byte(Src.Low);
auto GPRHigh = Mask4Byte(Src.High);
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 4);
} else {
GPR = Mask4Byte(_VUnZip2(OpSize::i128Bit, OpSize::i32Bit, Src.Low, Src.High));
auto GPRLow = Mask8Byte(Src.Low);
auto GPRHigh = Mask8Byte(Src.High);
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 2);
}
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
}
@@ -886,7 +885,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
auto Mask1Byte = [this](Ref Src, Ref VMask) {
auto VCMP = _VCMPLTZ(OpSize::i128Bit, OpSize::i8Bit, Src);
auto VAnd = _VAnd(OpSize::i128Bit, VCMP, VMask);
auto VAnd = _VAnd(OpSize::i128Bit, OpSize::i8Bit, VCMP, VMask);
auto VAdd1 = _VAddP(OpSize::i128Bit, OpSize::i8Bit, VAnd, VAnd);
auto VAdd2 = _VAddP(OpSize::i128Bit, OpSize::i8Bit, VAdd1, VAdd1);
@@ -1261,26 +1260,26 @@ void OpDispatchBuilder::AVX128_VAESKeyGenAssist(OpcodeArgs) {
}
void OpDispatchBuilder::AVX128_VPCMPESTRI(OpcodeArgs) {
PCMPXSTRXOpImpl(Op, true, false, true);
PCMPXSTRXOpImpl(Op, true, false);
///< Does not zero anything.
}
void OpDispatchBuilder::AVX128_VPCMPESTRM(OpcodeArgs) {
PCMPXSTRXOpImpl(Op, true, true, true);
PCMPXSTRXOpImpl(Op, true, true);
///< Zero the upper 128-bits of hardcoded YMM0
AVX128_StoreXMMRegister(0, LoadZeroVector(OpSize::i128Bit), true);
}
void OpDispatchBuilder::AVX128_VPCMPISTRI(OpcodeArgs) {
PCMPXSTRXOpImpl(Op, false, false, true);
PCMPXSTRXOpImpl(Op, false, false);
///< Does not zero anything.
}
void OpDispatchBuilder::AVX128_VPCMPISTRM(OpcodeArgs) {
PCMPXSTRXOpImpl(Op, false, true, true);
PCMPXSTRXOpImpl(Op, false, true);
///< Zero the upper 128-bits of hardcoded YMM0
AVX128_StoreXMMRegister(0, LoadZeroVector(OpSize::i128Bit), true);
@@ -1400,13 +1399,13 @@ void OpDispatchBuilder::AVX128_VSHUF(OpcodeArgs, IR::OpSize ElementSize) {
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
RefPair Result {};
Result.Low = SHUFOpImpl(OpSize::i128Bit, ElementSize, Src1.Low, Src2.Low, Shuffle);
Result.Low = SHUFOpImpl(Op, OpSize::i128Bit, ElementSize, Src1.Low, Src2.Low, Shuffle);
if (Is128Bit) {
Result.High = LoadZeroVector(OpSize::i128Bit);
} else {
const uint8_t ShiftAmount = ElementSize == OpSize::i32Bit ? 0 : 2;
Result.High = SHUFOpImpl(OpSize::i128Bit, ElementSize, Src1.High, Src2.High, Shuffle >> ShiftAmount);
Result.High = SHUFOpImpl(Op, OpSize::i128Bit, ElementSize, Src1.High, Src2.High, Shuffle >> ShiftAmount);
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
}
@@ -1470,35 +1469,6 @@ void OpDispatchBuilder::AVX128_VPMADDWD(OpcodeArgs) {
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return PMADDWDOpImpl(OpSize::i128Bit, Src1, Src2); });
}
void OpDispatchBuilder::AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper) {
const auto Size = OpSizeFromDst(Op);
const auto Is128Bit = Size == OpSize::i128Bit;
auto Acc = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit);
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
RefPair Result {};
Result.Low = Helper(Acc.Low, Src1.Low, Src2.Low);
if (Is128Bit) {
Result.High = LoadZeroVector(OpSize::i128Bit);
} else {
Result.High = Helper(Acc.High, Src1.High, Src2.High);
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
}
void OpDispatchBuilder::AVX128_VPDPBUSD(OpcodeArgs, bool Saturating) {
AVX128_VPDPImpl(Op,
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPBUSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
}
void OpDispatchBuilder::AVX128_VPDPWSSD(OpcodeArgs, bool Saturating) {
AVX128_VPDPImpl(Op,
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPWSSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
}
void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize) {
const auto SrcSize = OpSizeFromSrc(Op);
const auto Is128Bit = SrcSize == OpSize::i128Bit;
@@ -1514,12 +1484,12 @@ void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize) {
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
RefPair Result {};
Result.Low = VectorBlendImpl(OpSize::i128Bit, ElementSize, Src1.Low, Src2.Low, Selector);
Result.Low = VectorBlend(OpSize::i128Bit, ElementSize, Src1.Low, Src2.Low, Selector);
if (Is128Bit) {
Result = AVX128_Zext(Result.Low);
} else {
Result.High = VectorBlendImpl(OpSize::i128Bit, ElementSize, Src1.High, Src2.High, (Selector >> SelectorShift));
Result.High = VectorBlend(OpSize::i128Bit, ElementSize, Src1.High, Src2.High, (Selector >> SelectorShift));
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
@@ -1759,8 +1729,8 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize) {
{
// Calculate ZF first.
auto AndLow = _VAnd(OpSize::i128Bit, Src2.Low, Src1.Low);
auto AndHigh = _VAnd(OpSize::i128Bit, Src2.High, Src1.High);
auto AndLow = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low);
auto AndHigh = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High);
auto ShiftLow = _VUShrI(OpSize::i128Bit, ElementSize, AndLow, ElementSizeInBits - 1);
auto ShiftHigh = _VUShrI(OpSize::i128Bit, ElementSize, AndHigh, ElementSizeInBits - 1);
@@ -1779,8 +1749,8 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize) {
{
// Calculate CF Second
auto AndLow = _VAndn(OpSize::i128Bit, Src2.Low, Src1.Low);
auto AndHigh = _VAndn(OpSize::i128Bit, Src2.High, Src1.High);
auto AndLow = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low);
auto AndHigh = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High);
auto ShiftLow = _VUShrI(OpSize::i128Bit, ElementSize, AndLow, ElementSizeInBits - 1);
auto ShiftHigh = _VUShrI(OpSize::i128Bit, ElementSize, AndHigh, ElementSizeInBits - 1);
@@ -1818,11 +1788,11 @@ void OpDispatchBuilder::AVX128_PTest(OpcodeArgs) {
}
// For 256-bit, we need to unroll. This is nontrivial.
Ref Test1Low = _VAnd(OpSize::i128Bit, Src1.Low, Src2.Low);
Ref Test2Low = _VAndn(OpSize::i128Bit, Src2.Low, Src1.Low);
Ref Test1Low = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src1.Low, Src2.Low);
Ref Test2Low = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low);
Ref Test1High = _VAnd(OpSize::i128Bit, Src1.High, Src2.High);
Ref Test2High = _VAndn(OpSize::i128Bit, Src2.High, Src1.High);
Ref Test1High = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src1.High, Src2.High);
Ref Test2High = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High);
// Element size must be less than 32-bit for the sign bit tricks.
Ref Test1Max = _VUMax(OpSize::i128Bit, OpSize::i16Bit, Test1Low, Test1High);
@@ -2039,13 +2009,13 @@ void OpDispatchBuilder::AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Sr
ConstantEOR = LoadAndCacheNamedVectorConstant(
OpSize::i128Bit, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PSUBADDPS_INVERT : NAMED_VECTOR_PSUBADDPD_INVERT);
}
auto InvertedSourceLow = _VXor(OpSize::i128Bit, Sources[AddendIdx - 1].Low, ConstantEOR);
auto InvertedSourceLow = _VXor(OpSize::i128Bit, ElementSize, Sources[AddendIdx - 1].Low, ConstantEOR);
Result.Low = _VFMLA(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].Low, Sources[Src2Idx - 1].Low, InvertedSourceLow);
if (Is128Bit) {
Result.High = LoadZeroVector(OpSize::i128Bit);
} else {
auto InvertedSourceHigh = _VXor(OpSize::i128Bit, Sources[AddendIdx - 1].High, ConstantEOR);
auto InvertedSourceHigh = _VXor(OpSize::i128Bit, ElementSize, Sources[AddendIdx - 1].High, ConstantEOR);
Result.High = _VFMLA(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].High, Sources[Src2Idx - 1].High, InvertedSourceHigh);
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
@@ -2323,8 +2293,9 @@ void OpDispatchBuilder::AVX128_VCVTPS2PH(OpcodeArgs) {
_PopRoundingMode(OldFPCR);
}
// We need to zero the upper 128 bits if we're storing into a register
if (Op->Dest.IsGPR()) {
// We need to eliminate upper junk if we're storing into a register with
// a 256-bit source (VCVTPS2PH's destination for registers is an XMM).
if (Op->Src[0].IsGPR() && SrcSize == OpSize::i256Bit) {
Result = AVX128_Zext(Result.Low);
}
@@ -5,30 +5,21 @@
namespace FEXCore::IR {
constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
// Instructions
{0x00, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0, false>},
{0x04, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0, true>},
{0x00, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0>},
{0x08, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0, false>},
{0x0c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0, true>},
{0x08, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0>},
{0x10, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0, false>},
{0x14, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0, true>},
{0x10, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0>},
{0x18, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0, false>},
{0x1c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0, true>},
{0x18, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0>},
{0x20, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0, false>},
{0x24, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0, true>},
{0x20, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0>},
{0x28, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0, false>},
{0x2c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0, true>},
{0x28, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0>},
{0x30, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0, false>},
{0x34, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0, true>},
{0x38, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0, false>},
{0x3c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0, true>},
{0x30, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0>},
{0x38, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0>},
{0x50, 8, &OpDispatchBuilder::PUSHREGOp},
{0x58, 8, &OpDispatchBuilder::POPOp},
{0x68, 1, &OpDispatchBuilder::PUSHOp},
@@ -38,7 +29,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
{0x6C, 4, &OpDispatchBuilder::PermissionRestrictedOp},
{0x70, 16, &OpDispatchBuilder::CondJUMPOp},
{0x84, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0, false>},
{0x84, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
{0x86, 2, &OpDispatchBuilder::XCHGOp},
{0x88, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
@@ -46,7 +37,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
{0x8D, 1, &OpDispatchBuilder::LEAOp},
{0x8E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVSegOp, true>},
{0x8F, 1, &OpDispatchBuilder::POPOp},
{0x90, 8, &OpDispatchBuilder::XCHGRAXOp},
{0x90, 8, &OpDispatchBuilder::XCHGOp},
{0x98, 1, &OpDispatchBuilder::CDQOp},
{0x99, 1, &OpDispatchBuilder::CQOOp},
@@ -58,7 +49,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
{0xA4, 2, &OpDispatchBuilder::MOVSOp},
{0xA6, 2, &OpDispatchBuilder::CMPSOp},
{0xA8, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0, true>},
{0xA8, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
{0xAA, 2, &OpDispatchBuilder::STOSOp},
{0xAC, 2, &OpDispatchBuilder::LODSOp},
{0xAE, 2, &OpDispatchBuilder::SCASOp},
@@ -26,25 +26,17 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result {};
if (CTX->HostFeatures.SupportsSVE128) {
auto ZeroVec = LoadZeroVector(OpSize::i128Bit);
auto Tmp = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, ZeroVec, Dest);
auto Xar = _VXar(OpSize::i128Bit, OpSize::i32Bit, ZeroVec, Tmp, 2);
Result = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, Xar);
} else {
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
auto RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
}
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
auto RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
@@ -58,9 +50,9 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
Ref Result = _VXor(OpSize::i128Bit, Dest, NewVec);
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
@@ -78,7 +70,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
// The result is swizzled differently than expected
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
@@ -107,7 +99,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
break;
}
const auto ZeroRegister = LoadZeroVector(OpSize::i128Bit);
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
Ref Src1 = SHADataShuffle(Dest);
Ref Src2 = SHADataShuffle(Src);
@@ -120,7 +112,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
}
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
@@ -133,7 +125,7 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
auto Result = _VSha256U0(Dest, Src);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
@@ -150,7 +142,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
auto Result = _VSha256U1(Src1, Src2);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
@@ -185,22 +177,17 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
auto B = _VSha256H2(EFGH, ABCD, Key);
auto Result = shuffle_abcd(A, B);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::AESImcOp(OpcodeArgs, bool IsAVX) {
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESImc(Src);
if (IsAVX) {
StoreResultFPR(Op, Result);
} else {
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
@@ -211,30 +198,19 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is256Bit = DstSize == OpSize::i256Bit;
const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESENC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref ZeroVec = LoadZeroVector(DstSize);
Ref Result {};
if (Is256Bit) {
// TODO: Handle as one operation once vixl supports it.
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
auto Lower = _VAESEnc(OpSize::i128Bit, State, Key, ZeroVec);
auto Upper = _VAESEnc(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
} else {
Result = _VAESEnc(DstSize, State, Key, ZeroVec);
}
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
}
@@ -247,30 +223,19 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is256Bit = DstSize == OpSize::i256Bit;
const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESENCLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref ZeroVec = LoadZeroVector(DstSize);
Ref Result {};
if (Is256Bit) {
// TODO: Handle as one operation once vixl supports it.
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
auto Lower = _VAESEncLast(OpSize::i128Bit, State, Key, ZeroVec);
auto Upper = _VAESEncLast(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
} else {
Result = _VAESEncLast(DstSize, State, Key, ZeroVec);
}
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
}
@@ -283,30 +248,19 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is256Bit = DstSize == OpSize::i256Bit;
const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESDEC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref ZeroVec = LoadZeroVector(DstSize);
Ref Result {};
if (Is256Bit) {
// TODO: Handle as one operation once vixl supports it.
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
auto Lower = _VAESDec(OpSize::i128Bit, State, Key, ZeroVec);
auto Upper = _VAESDec(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
} else {
Result = _VAESDec(DstSize, State, Key, ZeroVec);
}
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
}
@@ -319,30 +273,19 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is256Bit = DstSize == OpSize::i256Bit;
const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESDECLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref ZeroVec = LoadZeroVector(DstSize);
Ref Result {};
if (Is256Bit) {
// TODO: Handle as one operation once vixl supports it.
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
auto Lower = _VAESDecLast(OpSize::i128Bit, State, Key, ZeroVec);
auto Upper = _VAESDecLast(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
} else {
Result = _VAESDecLast(DstSize, State, Key, ZeroVec);
}
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
StoreResultFPR(Op, Result);
}
@@ -355,19 +298,14 @@ Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
return _VAESKeyGenAssist(Src, KeyGenSwizzle, LoadZeroVector(OpSize::i128Bit), RCON);
}
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs, bool IsAVX) {
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Result = AESKeyGenAssistImpl(Op);
if (IsAVX) {
StoreResultFPR(Op, Result);
} else {
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
@@ -379,8 +317,8 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
auto Result = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
StoreResultFPR(Op, Res);
}
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
@@ -5,9 +5,9 @@
namespace FEXCore::IR {
constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
{0x0D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, false, false>},
{0x0D, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
{0x1D, 1, &OpDispatchBuilder::PF2IDOp},
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, false>},
@@ -15,27 +15,27 @@ constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
{0x90, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 1>},
{0x90, 1, &OpDispatchBuilder::VPFCMPOp<1>},
{0x94, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
{0x96, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryDuplicateOp, IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
{0x97, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, true>},
{0x9A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
{0x9E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 2>},
{0xA0, 1, &OpDispatchBuilder::VPFCMPOp<2>},
{0xA4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
// Can be treated as a move
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedNoNopOp},
{0xA7, 1, &OpDispatchBuilder::MOVVectorUnalignedNoNopOp},
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
{0xA7, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
{0xAA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VFSUB, OpSize::i32Bit>},
{0xAE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
{0xB0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 0>},
{0xB0, 1, &OpDispatchBuilder::VPFCMPOp<0>},
{0xB4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
// Can be treated as a move
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedNoNopOp},
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
{0xB7, 1, &OpDispatchBuilder::PMULHRWOp},
{0xBB, 1, &OpDispatchBuilder::PSWAPDOp},
@@ -432,7 +432,7 @@ void OpDispatchBuilder::CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res) {
SetNZP_ZeroCV(SrcSize, Res);
}
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift, bool DoubleWide) {
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
// No flags changed if shift is zero
if (Shift == 0) {
return;
@@ -447,12 +447,8 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Re
// Extract the last bit shifted in to CF. Shift is already masked, but for
// 8/16-bit it might be >= SrcSizeBits, in which case CF is cleared. There's
// nothing to do in that case since we already cleared CF above.
//
// - Double-wide shift has UB when shift is GREATER-THAN operand.
// - Single-wide shift has UB when shift is GREATER-THAN-EQUAL operand.
const auto SrcSizeBits = IR::OpSizeAsBits(SrcSize);
const bool ShouldSetCF = DoubleWide ? (Shift <= SrcSizeBits) : (Shift < SrcSizeBits);
if (ShouldSetCF) {
if (Shift < SrcSizeBits) {
SetCFDirect(Src1, SrcSizeBits - Shift, true);
}
}
@@ -20,18 +20,18 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
{OPD(PF_38_66, 0x03), 1, &OpDispatchBuilder::PHADDS},
{OPD(PF_38_NONE, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
{OPD(PF_38_66, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
{OPD(PF_38_NONE, 0x05), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PHSUB, OpSize::i16Bit>},
{OPD(PF_38_66, 0x05), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PHSUB, OpSize::i16Bit>},
{OPD(PF_38_NONE, 0x06), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PHSUB, OpSize::i32Bit>},
{OPD(PF_38_66, 0x06), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PHSUB, OpSize::i32Bit>},
{OPD(PF_38_NONE, 0x05), 1, &OpDispatchBuilder::PHSUB<OpSize::i16Bit>},
{OPD(PF_38_66, 0x05), 1, &OpDispatchBuilder::PHSUB<OpSize::i16Bit>},
{OPD(PF_38_NONE, 0x06), 1, &OpDispatchBuilder::PHSUB<OpSize::i32Bit>},
{OPD(PF_38_66, 0x06), 1, &OpDispatchBuilder::PHSUB<OpSize::i32Bit>},
{OPD(PF_38_NONE, 0x07), 1, &OpDispatchBuilder::PHSUBS},
{OPD(PF_38_66, 0x07), 1, &OpDispatchBuilder::PHSUBS},
{OPD(PF_38_NONE, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i8Bit>},
{OPD(PF_38_66, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i8Bit>},
{OPD(PF_38_NONE, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i16Bit>},
{OPD(PF_38_66, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i16Bit>},
{OPD(PF_38_NONE, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i32Bit>},
{OPD(PF_38_66, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i32Bit>},
{OPD(PF_38_NONE, 0x08), 1, &OpDispatchBuilder::PSIGN<OpSize::i8Bit>},
{OPD(PF_38_66, 0x08), 1, &OpDispatchBuilder::PSIGN<OpSize::i8Bit>},
{OPD(PF_38_NONE, 0x09), 1, &OpDispatchBuilder::PSIGN<OpSize::i16Bit>},
{OPD(PF_38_66, 0x09), 1, &OpDispatchBuilder::PSIGN<OpSize::i16Bit>},
{OPD(PF_38_NONE, 0x0A), 1, &OpDispatchBuilder::PSIGN<OpSize::i32Bit>},
{OPD(PF_38_66, 0x0A), 1, &OpDispatchBuilder::PSIGN<OpSize::i32Bit>},
{OPD(PF_38_NONE, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
{OPD(PF_38_66, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
{OPD(PF_38_66, 0x10), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, OpSize::i8Bit>},
@@ -44,22 +44,22 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i16Bit>},
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, true>},
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, true>},
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, true>},
{OPD(PF_38_66, 0x23), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, true>},
{OPD(PF_38_66, 0x24), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, true>},
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, true>},
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULLOp, OpSize::i32Bit, true>},
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, true>},
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, true>},
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, true>},
{OPD(PF_38_66, 0x23), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, true>},
{OPD(PF_38_66, 0x24), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, true>},
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, true>},
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, true>},
{OPD(PF_38_66, 0x29), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i64Bit>},
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKUSOp, OpSize::i32Bit>},
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, false>},
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, false>},
{OPD(PF_38_66, 0x32), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, false>},
{OPD(PF_38_66, 0x33), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, false>},
{OPD(PF_38_66, 0x34), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, false>},
{OPD(PF_38_66, 0x35), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, false>},
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::MOVVectorNTOp},
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::PACKUSOp<OpSize::i32Bit>},
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, false>},
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, false>},
{OPD(PF_38_66, 0x32), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, false>},
{OPD(PF_38_66, 0x33), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, false>},
{OPD(PF_38_66, 0x34), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, false>},
{OPD(PF_38_66, 0x35), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, false>},
{OPD(PF_38_66, 0x37), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i64Bit>},
{OPD(PF_38_66, 0x38), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i8Bit>},
{OPD(PF_38_66, 0x39), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i32Bit>},
@@ -79,7 +79,7 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
{OPD(PF_38_NONE, 0xCC), 1, &OpDispatchBuilder::SHA256MSG1Op},
{OPD(PF_38_NONE, 0xCD), 1, &OpDispatchBuilder::SHA256MSG2Op},
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESImcOp, false>},
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
{OPD(PF_38_66, 0xDC), 1, &OpDispatchBuilder::AESEncOp},
{OPD(PF_38_66, 0xDD), 1, &OpDispatchBuilder::AESEncLastOp},
{OPD(PF_38_66, 0xDE), 1, &OpDispatchBuilder::AESDecOp},
@@ -9,13 +9,13 @@ namespace FEXCore::IR {
constexpr auto OpDispatchTableGenH0F3A = []() consteval {
constexpr auto OpDispatchTableGenH0F3AREX = []<uint16_t REX>() consteval {
constexpr DispatchTableEntry Table[] = {
{OPD(REX, PF_3A_66, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorRound, OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorRound, OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalarRound, OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalarRound, OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorBlend, OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorBlend, OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorBlend, OpSize::i16Bit>},
{OPD(REX, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<OpSize::i16Bit>},
{OPD(REX, PF_3A_NONE, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
{OPD(REX, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
@@ -24,20 +24,20 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
{OPD(REX, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
{OPD(REX, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x20), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PINSROp, OpSize::i8Bit>},
{OPD(REX, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<OpSize::i8Bit>},
{OPD(REX, PF_3A_66, 0x21), 1, &OpDispatchBuilder::InsertPSOp},
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::DPPOp, OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::DPPOp, OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
{OPD(REX, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPESTRMOp, false>},
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPESTRIOp, false>},
{OPD(REX, PF_3A_66, 0x62), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRMOp, false>},
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRIOp, false>},
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
{OPD(REX, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESKeyGenAssist, false>},
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
};
return std::to_array(Table);
@@ -65,7 +65,7 @@ constexpr auto OpDispatch_H0F3ATableIgnoreREX = OpDispatchTableGenH0F3A();
constexpr DispatchTableEntry OpDispatch_H0F3ATableNeedsREX0[] = {
{OPD(0, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PINSROp, OpSize::i32Bit>},
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i32Bit>},
};
#undef PF_3A_NONE
@@ -9,36 +9,36 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
// GROUP 1
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>}, // CMP
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
// GROUP 2
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLOp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCROp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 2), 1, &OpDispatchBuilder::RCLOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 3), 1, &OpDispatchBuilder::RCROp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
@@ -46,8 +46,8 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLOp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCROp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 2), 1, &OpDispatchBuilder::RCLOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 3), 1, &OpDispatchBuilder::RCROp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
@@ -73,8 +73,8 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLSmallerOp, true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCRSmallerOp, true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, &OpDispatchBuilder::RCLSmallerOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, &OpDispatchBuilder::RCRSmallerOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, &OpDispatchBuilder::SHLOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
@@ -82,16 +82,16 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLOp, true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCROp, true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, &OpDispatchBuilder::RCLOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, &OpDispatchBuilder::RCROp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, &OpDispatchBuilder::SHLOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, false, false>}, // SAR
// GROUP 3
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 2), 1, &OpDispatchBuilder::NOTOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 4), 1, &OpDispatchBuilder::MULOp},
@@ -99,8 +99,8 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 6), 1, &OpDispatchBuilder::DIVOp}, // DIV
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 7), 1, &OpDispatchBuilder::IDIVOp}, // IDIV
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, &OpDispatchBuilder::NOTOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
@@ -69,12 +69,12 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
// GROUP 9
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RDRANDOp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RDRANDOp, true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RDRANDOp, false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RDRANDOp, true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F2, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
@@ -6,7 +6,8 @@ namespace FEXCore::IR {
constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
// Instructions
{0x03, 1, &OpDispatchBuilder::LSLOp},
{0x06, 4, &OpDispatchBuilder::PermissionRestrictedOp},
{0x06, 1, &OpDispatchBuilder::PermissionRestrictedOp},
{0x07, 1, &OpDispatchBuilder::PermissionRestrictedOp},
{0x0B, 1, &OpDispatchBuilder::INTOp},
{0x0E, 1, &OpDispatchBuilder::X87EMMS},
@@ -43,7 +44,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
{0xBE, 2, &OpDispatchBuilder::MOVSXOp},
{0xC0, 2, &OpDispatchBuilder::XADDOp},
{0xC3, 1, &OpDispatchBuilder::MOVGPRNTOp},
{0xC4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PINSROp, OpSize::i16Bit>},
{0xC4, 1, &OpDispatchBuilder::PINSROp<OpSize::i16Bit>},
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
{0xC8, 8, &OpDispatchBuilder::BSWAPOp},
@@ -55,10 +56,10 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
{0x2A, 1, &OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float},
{0x2B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
{0x2C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int, OpSize::i32Bit, false>},
{0x2D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int, OpSize::i32Bit, true>},
{0x2E, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i32Bit>},
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i32Bit>},
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i32Bit>},
{0x52, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
@@ -70,7 +71,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit, false>},
{0x5B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, false, false>},
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i32Bit>},
@@ -78,15 +79,15 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i8Bit>},
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i16Bit>},
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
{0x63, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKSSOp, OpSize::i16Bit>},
{0x63, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i16Bit>},
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i8Bit>},
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i16Bit>},
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i32Bit>},
{0x67, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKUSOp, OpSize::i16Bit>},
{0x67, 1, &OpDispatchBuilder::PACKUSOp<OpSize::i16Bit>},
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i8Bit>},
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i16Bit>},
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
{0x6B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKSSOp, OpSize::i32Bit>},
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i32Bit>},
{0x70, 1, &OpDispatchBuilder::PSHUFW8ByteOp},
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i8Bit>},
@@ -94,7 +95,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
{0x77, 1, &OpDispatchBuilder::X87EMMS},
{0xC2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFCMPOp, OpSize::i32Bit>},
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<OpSize::i32Bit>},
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, OpSize::i32Bit>},
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i16Bit>},
@@ -115,9 +116,9 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i16Bit>},
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i32Bit>},
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
{0xE4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULHW, false>},
{0xE5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULHW, true>},
{0xE7, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i16Bit>},
@@ -130,7 +131,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i16Bit>},
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i32Bit>},
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i64Bit>},
{0xF4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULLOp, OpSize::i32Bit, false>},
{0xF4, 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, false>},
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
{0xF6, 1, &OpDispatchBuilder::PSADBW},
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
@@ -151,23 +152,23 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
constexpr DispatchTableEntry OpDispatch_SecondaryRepModTables[] = {
{0x10, 2, &OpDispatchBuilder::MOVSSOp},
{0x12, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMOVSLDUPOp, false>},
{0x16, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMOVSHDUPOp, false>},
{0x2A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertCVTGPR_To_FPR, OpSize::i32Bit>},
{0x2B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
{0x2C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i32Bit, false>},
{0x2D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i32Bit, true>},
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarUnaryInsertALUOp, IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
{0x52, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarUnaryInsertALUOp, IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
{0x53, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarUnaryInsertALUOp, IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalar_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit>},
{0x5B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, false, false>},
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
{0x12, 1, &OpDispatchBuilder::VMOVSLDUPOp},
{0x16, 1, &OpDispatchBuilder::VMOVSHDUPOp},
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<OpSize::i32Bit>},
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, false>},
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, true>},
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
{0x52, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
{0x53, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
{0x6F, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, false>},
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQOp, OpDispatchBuilder::VectorOpType::SSE>},
@@ -175,36 +176,36 @@ constexpr DispatchTableEntry OpDispatch_SecondaryRepModTables[] = {
{0xB8, 1, &OpDispatchBuilder::PopcountOp},
{0xBC, 1, &OpDispatchBuilder::TZCNT},
{0xBD, 1, &OpDispatchBuilder::LZCNT},
{0xC2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalarFCMPOp, OpSize::i32Bit>},
{0xD6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQ2DQ, true>},
{0xE6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, true, false>},
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i32Bit>},
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<true>},
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
};
constexpr DispatchTableEntry OpDispatch_SecondaryRepNEModTables[] = {
{0x10, 2, &OpDispatchBuilder::MOVSDOp},
{0x12, 1, &OpDispatchBuilder::MOVDDUPOp},
{0x2A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertCVTGPR_To_FPR, OpSize::i64Bit>},
{0x2B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
{0x2C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i64Bit, false>},
{0x2D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i64Bit, true>},
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarUnaryInsertALUOp, IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<OpSize::i64Bit>},
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, false>},
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, true>},
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
// x52 = Invalid
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalar_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit>},
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
{0x78, 1, &OpDispatchBuilder::Insertq_imm},
{0x79, 1, &OpDispatchBuilder::Insertq},
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
{0x7D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::HSUBP, OpSize::i32Bit>},
{0xD0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADDSUBPOp, OpSize::i32Bit>},
{0xD6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQ2DQ, false>},
{0xC2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalarFCMPOp, OpSize::i64Bit>},
{0xE6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i64Bit, true, false>},
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i32Bit>},
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<false>},
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i64Bit>},
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
{0xF0, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
};
@@ -216,10 +217,10 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float},
{0x2B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
{0x2C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int, OpSize::i64Bit, false>},
{0x2D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int, OpSize::i64Bit, true>},
{0x2E, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i64Bit>},
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i64Bit>},
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i64Bit>},
@@ -230,7 +231,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i64Bit>},
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i64Bit>},
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit, false>},
{0x5B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, true, false>},
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i64Bit>},
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i64Bit>},
@@ -238,15 +239,15 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i8Bit>},
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i16Bit>},
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
{0x63, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKSSOp, OpSize::i16Bit>},
{0x63, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i16Bit>},
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i8Bit>},
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i16Bit>},
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i32Bit>},
{0x67, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKUSOp, OpSize::i16Bit>},
{0x67, 1, &OpDispatchBuilder::PACKUSOp<OpSize::i16Bit>},
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i8Bit>},
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i16Bit>},
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
{0x6B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKSSOp, OpSize::i32Bit>},
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i32Bit>},
{0x6C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i64Bit>},
{0x6D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i64Bit>},
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
@@ -259,15 +260,15 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
{0x78, 1, nullptr}, // GROUP 17
{0x79, 1, &OpDispatchBuilder::Extrq},
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>},
{0x7D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::HSUBP, OpSize::i64Bit>},
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i64Bit>},
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
{0x7F, 1, &OpDispatchBuilder::MOVVectorAlignedOp},
{0xC2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFCMPOp, OpSize::i64Bit>},
{0xC4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PINSROp, OpSize::i16Bit>},
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<OpSize::i64Bit>},
{0xC4, 1, &OpDispatchBuilder::PINSROp<OpSize::i16Bit>},
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, OpSize::i64Bit>},
{0xD0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADDSUBPOp, OpSize::i64Bit>},
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i64Bit>},
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i16Bit>},
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i32Bit>},
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i64Bit>},
@@ -287,10 +288,10 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i16Bit>},
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i32Bit>},
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
{0xE4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULHW, false>},
{0xE5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULHW, true>},
{0xE6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i64Bit, false, false>},
{0xE7, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i16Bit>},
@@ -303,7 +304,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i16Bit>},
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i32Bit>},
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i64Bit>},
{0xF4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULLOp, OpSize::i32Bit, false>},
{0xF4, 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, false>},
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
{0xF6, 1, &OpDispatchBuilder::PSADBW},
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
File diff suppressed because it is too large. Load diff
@@ -139,7 +139,9 @@ void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
void OpDispatchBuilder::FSTToStack(OpcodeArgs) {
const uint8_t Offset = Op->OP & 7;
_StoreStackToStack(Offset);
if (Offset != 0) {
_StoreStackToStack(Offset);
}
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
_PopStackDestroy();
@@ -170,9 +172,8 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
Ref IsOverflow = _NZCVSelect01(CondClass::UGE);
// Set Invalid Operation flag if overflow or special value
// The x87 exception flags are sticky. Preserve earlier result
Ref InvalidFlag = _Or(OpSize::i64Bit, IsSpecial, IsOverflow);
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Or(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::X87FLAG_IE_LOC), InvalidFlag));
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(InvalidFlag);
}
Data = _F80CVTInt(Size, Data, Truncate);
@@ -574,7 +575,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
for (int i = 0; i < 7; ++i) {
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
// Mask off the top bits
Reg = _VAnd(OpSize::i128Bit, Reg, Mask);
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
if (ReducedPrecisionMode) {
// Convert to double precision
Reg = _F80CVT(OpSize::i64Bit, Reg);
@@ -623,10 +624,13 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
void OpDispatchBuilder::X87FYL2X(OpcodeArgs, bool IsFYL2XP1) {
if (IsFYL2XP1) {
_F80FYL2XP1Stack();
} else {
_F80FYL2XStack();
// create an add between top of stack and 1.
Ref One = ReducedPrecisionMode ? _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0x3FF0000000000000)) :
LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
_F80AddValue(0, One);
}
_F80FYL2XStack();
}
void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
@@ -666,6 +670,7 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
} else {
// OF, SF, AF, PF all undefined
SetCFDirect(HostFlag_CF);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
@@ -673,17 +678,10 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
// TODO: This could perhaps be optimized?
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
// Intel: OF, SF, and AF set to zero
// AMD: no mention of OF, SF and AF but actual hardware seems to always zero
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Constant(0));
SetRFLAG<FEXCore::X86State::RFLAG_SF_RAW_LOC>(Constant(0));
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
}
// Set Invalid Operation flag when unordered (NaN comparison)
// The x87 exception flags are sticky. Preserve earlier result
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Or(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::X87FLAG_IE_LOC), HostFlag_Unordered));
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(HostFlag_Unordered);
if (PopTwice) {
_PopStackDestroy();
@@ -708,8 +706,7 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
// Set Invalid Operation flag when unordered (NaN comparison)
// The x87 exception flags are sticky. Preserve earlier result
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Or(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::X87FLAG_IE_LOC), HostFlag_Unordered));
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(HostFlag_Unordered);
}
void OpDispatchBuilder::X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2) {
@@ -725,9 +722,6 @@ void OpDispatchBuilder::X87ModifySTP(OpcodeArgs, bool Inc) {
} else {
_DecStackTop();
}
// C1 set to 0
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
}
// Operations dealing with loading and storing environment pieces
@@ -768,14 +762,10 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
// Store Status Word
// There's no load Status Word instruction but you can load it through frstor
// or fldenv.
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs, bool DestRAX) {
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
Ref TopValue = _SyncStackToSlow();
Ref StatusWord = ReconstructFSW_Helper(TopValue);
if (DestRAX) {
StoreGPRRegister(X86State::REG_RAX, StatusWord, OpSize::i16Bit);
} else {
StoreResultGPR(Op, StatusWord);
}
StoreResultGPR(Op, StatusWord);
}
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
@@ -860,103 +850,22 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
auto a = _ReadStackValue(0);
Ref Value = ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
Ref Result =
ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
// Extract the sign bit, which goes in C1
Ref Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Value) : _Bfe(OpSize::i64Bit, 1, 15, Value);
// Extract the sign bit
Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Result) : _Bfe(OpSize::i64Bit, 1, 15, Result);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
auto NotEmpty = _StackValidTag(0);
Ref IsEmpty = _Xor(OpSize::i64Bit, NotEmpty, Constant(1));
Ref IsNaN {};
Ref IsDenormal {};
Ref IsInf {};
Ref IsZero {};
Ref IsUnsupported {};
Ref NoSignBit {};
// Claim this is a normal number
// We don't support anything else
auto TopValid = _StackValidTag(0);
// TODO: The codegen for this is not optimal, and can probably be improved
// if FXAM ends up on the hot path for some workload.
if (ReducedPrecisionMode) {
constexpr uint64_t ExponentMask = 0x7FF0'0000'0000'0000ULL;
NoSignBit = _Bfe(OpSize::i64Bit, 63, 0, Value);
IsInf = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(ExponentMask));
IsNaN = Select01(OpSize::i64Bit, CondClass::UGT, NoSignBit, Constant(ExponentMask));
IsZero = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(0));
// 64 bit floats can't represent an x87 denormal, nor any of the
// unsupported encodings.
IsDenormal = Constant(0);
IsUnsupported = Constant(0);
} else {
Ref Mantissa = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 0);
// "J" is the name given to the msb of the mantissa in the SDM.
Ref JBit = _Bfe(OpSize::i64Bit, 1, 63, Mantissa);
Ref Exponent = _Bfe(OpSize::i64Bit, 15, 0, Value);
Ref IsExponentZero = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0));
Ref IsExponentMax = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0x7FFF));
// Inf is when mantissa only has the J bit set, exponent is all 1's.
Ref IsOnlyJBit = Select01(OpSize::i64Bit, CondClass::EQ, Mantissa, Constant(1ULL << 63));
IsInf = _And(OpSize::i64Bit, IsExponentMax, IsOnlyJBit);
// NaN is when the low 63 bits of the mantissa are non-zero
// and exponent is max, and the J bit is set.
Ref Fraction = _Bfe(OpSize::i64Bit, 63, 0, Mantissa);
Ref FractionNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Fraction, Constant(0));
Ref IsExponentMaxWithJBit = _And(OpSize::i64Bit, IsExponentMax, JBit);
IsNaN = _And(OpSize::i64Bit, IsExponentMaxWithJBit, FractionNonZero);
// Zero and Denormal are basically the same as the 64-bit case.
Ref MantissaNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Mantissa, Constant(0));
Ref MantissaZero = _Xor(OpSize::i64Bit, MantissaNonZero, Constant(1));
IsZero = _And(OpSize::i64Bit, IsExponentZero, MantissaZero);
IsDenormal = _And(OpSize::i64Bit, IsExponentZero, MantissaNonZero);
// This is where things are weird. If the J bit is not set
// and the exponent is non-zero, then this is an "unsupported"
// encoding, which I believe is left in for legacy reasons.
Ref IsSupported = _Or(OpSize::i64Bit, IsExponentZero, JBit);
IsUnsupported = _Xor(OpSize::i64Bit, IsSupported, Constant(1));
}
// NormalFiniteNumber = !Zero && !Denormal && !Inf && !NaN && !Empty && !Unsupported
Ref temp1 = _Or(OpSize::i64Bit, IsZero, IsDenormal);
Ref temp2 = _Or(OpSize::i64Bit, IsInf, IsNaN);
Ref temp3 = _Or(OpSize::i64Bit, IsUnsupported, IsEmpty);
temp1 = _Or(OpSize::i64Bit, temp1, temp2);
temp2 = _Or(OpSize::i64Bit, temp1, temp3);
Ref NormalFiniteNumber = _Xor(OpSize::i64Bit, temp2, Constant(1));
// Set C3, C2, C0 based on the class of the FP value
// Table is from "FXAM" page in the SDM.
// +----------------------+----+----+----+
// | Class | C3 | C2 | C0 |
// +----------------------+----+----+----+
// | Unsupported | 0 | 0 | 0 |
// | NaN | 0 | 0 | 1 |
// | Normal finite number | 0 | 1 | 0 |
// | Infinity | 0 | 1 | 1 |
// | Zero | 1 | 0 | 0 |
// | Empty | 1 | 0 | 1 |
// | Denormal number | 1 | 1 | 0 |
// +----------------------+----+----+----+
// C0 = IsNaN || IsInf || IsEmpty
Ref C0 = _Or(OpSize::i64Bit, IsNaN, IsInf);
C0 = _Or(OpSize::i64Bit, C0, IsEmpty);
// C2 = (IsInf || Denormal || NormalFiniteNumber) && !IsEmpty
Ref C2 = _Or(OpSize::i64Bit, IsInf, IsDenormal);
C2 = _Or(OpSize::i64Bit, C2, NormalFiniteNumber);
C2 = _And(OpSize::i64Bit, C2, NotEmpty);
// C3 = Zero || IsEmpty || Denormal
Ref C3 = _Or(OpSize::i64Bit, IsZero, IsEmpty);
C3 = _Or(OpSize::i64Bit, C3, IsDenormal);
// In the case of top being invalid then C3:C2:C0 is 0b101
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
auto C2 = TopValid;
auto C0 = C3; // Mirror C3 until something other than zero is supported
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(C0);
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(C2);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(C3);
@@ -367,7 +367,7 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
} else {
HandleNZCVWrite();
_F80CmpValue(b);
ComissFlags();
ComissFlags(true /* InvalidateAF */);
}
if (PopTwice) {
@@ -1,112 +0,0 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/SharedCodeBufferManager.h"
#include <FEXCore/fextl/memory.h>
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#ifndef _WIN32
#include <FEXCore/Utils/PrctlUtils.h>
#endif
namespace FEXCore::CPU {
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
CodeBuffer::CodeBuffer(size_t Size, bool ShouldBeNamed)
: AllocatedSize(Size) {
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
FEXCore::Allocator::ProtectOptions::None)) {
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
}
if (ShouldBeNamed) {
FEXCore::Allocator::VirtualName("FEXMemJIT", Ptr, Size);
}
// Huge-pages reduce the amount of iTLB misses dramatically when it works.
FEXCore::Allocator::VirtualTHPControl(Ptr, Size, FEXCore::Allocator::THPControl::Enable);
LookupCache = fextl::make_unique<GuestToHostMap>();
CodeBufferEnd = Ptr + UsableSize();
CodeBufferOffset = Ptr;
}
CodeBuffer::~CodeBuffer() {
FEXCore::Allocator::VirtualFree(Ptr, AllocatedSize);
}
SharedCodeBufferManager::SharedCodeBufferManager() {
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
// Only name the JIT buffers if perf JIT naming is disabled.
// `perf top` prefers VMA names over the JIT symbols file for some reason.
// Breaks memory tracking when naming is enabled, but it's a debug feature so it isn't expected to be enabled by default.
NameJITBuffers = !(GlobalJITNaming || LibraryJITNaming || BlockJITNaming);
}
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::AllocateNew(size_t Size) {
#ifndef _WIN32
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
//
// MDWE prevents applications from creating RWX memory mappings.
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
//
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
//
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
//
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
// -1: The kernel doesn't support MDWE
// 0: MDWE is supported but disabled
// >0: MDWE is enabled, hence prohibiting RWX mappings
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
if (MDWE != -1 && MDWE != 0) {
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
}
#endif
auto Buffer = fextl::make_shared<CodeBuffer>(Size, NameJITBuffers);
Latest = Buffer;
OnCodeBufferAllocated(Buffer);
return Buffer;
}
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::GetLatest() {
if (!Latest) {
AllocateNew(INITIAL_CODE_SIZE);
}
return Latest;
}
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::StartLargerCodeBuffer() {
if (!Latest) {
// Allocate initial CodeBuffer and return it
return GetLatest();
}
auto NewCodeBufferSize = GetLatest()->TotalAllocationSize();
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
return AllocateNew(NewCodeBufferSize);
}
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::StartMaximalCodeBuffer() {
return AllocateNew(MAX_CODE_SIZE);
}
} // namespace FEXCore::CPU
@@ -1,138 +0,0 @@
// SPDX-License-Identifier: MIT
/*
$info$
category: code buffer ~ Thread shared code buffer management
tags: backend|shared
$end_info$
*/
#pragma once
#include <FEXCore/fextl/memory.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <cstddef>
#include <cstdint>
namespace FEXCore {
struct GuestToHostMap;
}
namespace FEXCore::CPU {
struct CodeBuffer {
fextl::unique_ptr<GuestToHostMap> LookupCache;
CodeBuffer(size_t Size, bool ShouldBeNamed);
CodeBuffer(const CodeBuffer&) = delete;
CodeBuffer& operator=(const CodeBuffer&) = delete;
CodeBuffer(CodeBuffer&& oth) = delete;
CodeBuffer& operator=(CodeBuffer&&) = delete;
~CodeBuffer();
// Atomically allocate a fixed size buffer out of the current allocated codebuffer.
// Lockless because it's just a linear allocator.
struct CodeBufferAllocation {
const uint8_t* BufferBase;
uint8_t* BufferAllocationOffset;
};
CodeBufferAllocation AtomicAllocateBuffer(size_t Size) {
Size = FEXCore::AlignUp(Size, 16);
LOGMAN_THROW_A_FMT(reinterpret_cast<uintptr_t>(CodeBufferOffset.load()) % 16 == 0, "Buffer needs to always be 16B aligned!");
auto ExpectedOffset = CodeBufferOffset.load(std::memory_order_relaxed);
auto DesiredOffset = ExpectedOffset + Size;
if (DesiredOffset > CodeBufferEnd) {
// Couldn't fit.
return {};
}
while (!CodeBufferOffset.compare_exchange_strong(ExpectedOffset, DesiredOffset)) {
DesiredOffset = ExpectedOffset + Size;
if (DesiredOffset > CodeBufferEnd) {
// Couldn't fit.
return {};
}
}
// Managed to fit.
return {
.BufferBase = Ptr,
.BufferAllocationOffset = ExpectedOffset,
};
}
// Returns the total number of bytes available for storing code
size_t UsableSize() const {
return AllocatedSize - FEXCore::Utils::FEX_PAGE_SIZE;
}
// Returns the full size of the buffer, including the guard page.
size_t TotalAllocationSize() const {
return AllocatedSize;
}
// Returns the num of bytes currently allocated from the allocator.
size_t AllocatedSpaceUsed() const {
return CodeBufferOffset - Ptr;
}
// Trivially reset the allocator.
void Reset() {
CodeBufferOffset = Ptr;
}
// Returns the base of the buffer.
uint8_t* GetBufferBase() const {
return Ptr;
}
private:
uint8_t* Ptr;
uint8_t* CodeBufferEnd;
size_t AllocatedSize; // including guard page; see UsableSize()
// Code buffer allocation information.
std::atomic<uint8_t*> CodeBufferOffset {};
};
/**
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
*
* The CodeBuffer is managed as a partially persistent data structure:
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
*/
class SharedCodeBufferManager {
public:
SharedCodeBufferManager();
virtual ~SharedCodeBufferManager() = default;
// Get the CodeBuffer that was most recently allocated.
// This is the only CodeBuffer that data may be written to.
fextl::shared_ptr<CodeBuffer> GetLatest();
// Allocate a new CodeBuffer with geometric growth up to an internal maximum.
// Subsequent calls to GetLatest will point to the returned buffer.
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer();
// Allocate a new CodeBuffer with maximum internal size.
// Subsequent calls to GetLatest will point to the returned buffer.
fextl::shared_ptr<CodeBuffer> StartMaximalCodeBuffer();
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
private:
fextl::shared_ptr<CodeBuffer> Latest;
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
bool NameJITBuffers {true};
};
} // namespace FEXCore::CPU
@@ -83,22 +83,22 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
},
// ENTRY_27
{
{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::DAAOp } },
{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::DAAOp } },
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
// ENTRY_2F
{
{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::DASOp } },
{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::DASOp } },
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
// ENTRY_37
{
{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::AAAOp } },
{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::AAAOp } },
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
// ENTRY_3F
{
{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::AASOp } },
{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::AASOp } },
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
// ENTRY_40
@@ -134,23 +134,23 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
},
// ENTRY_A0
{
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
},
// ENTRY_A1
{
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
},
// ENTRY_A2
{
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
},
// ENTRY_A3
{
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
},
// ENTRY_CE
{
@@ -159,17 +159,17 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
},
// ENTRY_D4
{
{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1, { .OpDispatch = &IR::OpDispatchBuilder::AAMOp } },
{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, { .OpDispatch = &IR::OpDispatchBuilder::AAMOp } },
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
// ENTRY_D5
{
{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1, { .OpDispatch = &IR::OpDispatchBuilder::AADOp } },
{"REX2", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, { .OpDispatch = &IR::OpDispatchBuilder::AADOp } },
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
// ENTRY_D6
{
{"SALC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::SALCOp } },
{"SALC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::SALCOp } },
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
// ENTRY_EA
@@ -200,72 +200,72 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
{0xF3, 1, X86InstInfo{"REP", TYPE_PREFIX, FLAGS_NONE, 0}},
// Instructions
{0x00, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x01, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
{0x00, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x01, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0}},
{0x02, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
{0x03, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM, 0}},
{0x04, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
{0x05, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0x04, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
{0x05, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{0x06, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_06] }}},
{0x07, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_07] }}},
{0x08, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x09, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x08, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x09, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x0A, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
{0x0B, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM, 0}},
{0x0C, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
{0x0D, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0x0C, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
{0x0D, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{0x0E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_0E] }}},
{0x10, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x11, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
{0x10, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x11, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0}},
{0x12, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
{0x13, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM, 0}},
{0x14, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
{0x15, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0x14, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
{0x15, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{0x16, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_16] }}},
{0x17, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_17] }}},
{0x18, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x19, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
{0x18, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x19, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0}},
{0x1A, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
{0x1B, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM, 0}},
{0x1C, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
{0x1D, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0x1C, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
{0x1D, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{0x1E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1E] }}},
{0x1F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1F] }}},
{0x20, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x21, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x20, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x21, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x22, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
{0x23, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM, 0}},
{0x24, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
{0x25, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0x24, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
{0x25, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{0x27, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_27] }}},
{0x28, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x29, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x28, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x29, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x2A, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
{0x2B, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM, 0}},
{0x2C, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
{0x2D, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0x2C, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
{0x2D, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{0x2F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_2F] }}},
{0x30, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x31, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x30, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x31, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x32, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
{0x33, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM, 0}},
{0x34, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
{0x35, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0x34, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
{0x35, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{0x37, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_37] }}},
{0x38, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x39, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x3A, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
{0x3B, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM, 0}},
{0x3C, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
{0x3D, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0x3C, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
{0x3D, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{0x3F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_3F] }}},
{0x40, 8, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_40] }}},
@@ -310,20 +310,20 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
{0x84, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x85, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x86, 1, X86InstInfo{"XCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x87, 1, X86InstInfo{"XCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0x86, 1, X86InstInfo{"XCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x87, 1, X86InstInfo{"XCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x88, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x89, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x8A, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
{0x8B, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM, 0}},
{0x8C, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0x8D, 1, X86InstInfo{"LEA", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
{0x8D, 1, X86InstInfo{"LEA", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM, 0}},
{0x8E, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_16BIT) | FLAGS_MODRM, 0}},
{0x8F, 1, X86InstInfo{"POP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_ZERO_REG | FLAGS_DEBUG_MEM_ACCESS, 0}},
{0x90, 8, X86InstInfo{"XCHG", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0}},
{0x98, 1, X86InstInfo{"CDQE", TYPE_INST, FLAGS_NONE, 0}},
{0x99, 1, X86InstInfo{"CQO", TYPE_INST, FLAGS_NONE, 0}},
{0x90, 8, X86InstInfo{"XCHG", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_SF_SRC_RAX, 0}},
{0x98, 1, X86InstInfo{"CDQE", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0}},
{0x99, 1, X86InstInfo{"CQO", TYPE_INST, FLAGS_SF_DST_RDX | FLAGS_SF_SRC_RAX, 0}},
// These three are all X87 instructions
{0x9B, 1, X86InstInfo{"FWAIT", TYPE_INST, FLAGS_NONE, 0}},
@@ -343,17 +343,17 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
{0xB0, 8, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_REX_IN_BYTE , 1}},
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2, 4}},
{0xC2, 1, X86InstInfo{"RET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2}},
{0xC3, 1, X86InstInfo{"RET", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END , 0}},
@@ -362,7 +362,7 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
{0xCA, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2}},
{0xCB, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_BLOCK_END, 0}},
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 1}},
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1}},
{0xCE, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_CE] }}},
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
@@ -371,9 +371,9 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
{0xD6, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_D6] }}},
{0xD7, 1, X86InstInfo{"XLAT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
{0xE0, 1, X86InstInfo{"LOOPNE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT, 1}},
{0xE1, 1, X86InstInfo{"LOOPE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT, 1}},
{0xE2, 1, X86InstInfo{"LOOP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT, 1}},
{0xE0, 1, X86InstInfo{"LOOPNE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
{0xE1, 1, X86InstInfo{"LOOPE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
{0xE2, 1, X86InstInfo{"LOOP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
{0xE3, 1, X86InstInfo{"JrCXZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
// Should just throw GP
@@ -34,7 +34,7 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> H0F3A_ArchSelect_LUT = {{
// ENTRY_1_3A_66_22
{
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
{"PINSRQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PINSROp, IR::OpSize::i64Bit> }},
{"PINSRQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, { .OpDispatch = &IR::OpDispatchBuilder::PINSROp<IR::OpSize::i64Bit> }},
},
}};
@@ -28,35 +28,35 @@ enum PrimaryGroup_LUT {
constexpr std::array<X86InstInfo[2], ENTRY_MAX> PrimaryGroup_ArchSelect_LUT = {{
{
{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
{
{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
{
{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::ADCOp, 1, false> }},
{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::ADCOp, 1> }},
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
{
{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SBBOp, 1, false> }},
{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SBBOp, 1> }},
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
{
{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
{
{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
{
{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
{
{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::CMPOp, 1, false> }},
{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::CMPOp, 1> }},
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
},
}};
@@ -66,23 +66,23 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
constexpr U16U8InfoStruct PrimaryGroupOpTable[] = {
// GROUP_1 | 0x80 | reg
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 0), 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 1), 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 2), 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 3), 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 4), 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 5), 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 6), 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 7), 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 0), 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 1), 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 2), 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 3), 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 4), 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 5), 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 6), 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 7), 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
// Duplicates the 0x80 opcode group
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 0), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = PrimaryGroup_ArchSelect_LUT[ENTRY_1_82_0] }}},
@@ -94,14 +94,14 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 6), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = PrimaryGroup_ArchSelect_LUT[ENTRY_1_82_6] }}},
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 7), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = PrimaryGroup_ArchSelect_LUT[ENTRY_1_82_7] }}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
// GROUP 2
{OPD(TYPE_GROUP_2, OpToIndex(0xC0), 0), 1, X86InstInfo{"ROL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
@@ -140,51 +140,51 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
{OPD(TYPE_GROUP_2, OpToIndex(0xD1), 6), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD1), 7), 1, X86InstInfo{"SAR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, X86InstInfo{"ROL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, X86InstInfo{"ROR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, X86InstInfo{"RCL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, X86InstInfo{"RCR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, X86InstInfo{"SHR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, X86InstInfo{"SAR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, X86InstInfo{"ROL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, X86InstInfo{"ROR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, X86InstInfo{"RCL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, X86InstInfo{"RCR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, X86InstInfo{"SHR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, X86InstInfo{"SAR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, X86InstInfo{"ROL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, X86InstInfo{"ROR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, X86InstInfo{"RCL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, X86InstInfo{"RCR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, X86InstInfo{"SHR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, X86InstInfo{"SAR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, X86InstInfo{"ROL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, X86InstInfo{"ROR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, X86InstInfo{"RCL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, X86InstInfo{"RCR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, X86InstInfo{"SHR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, X86InstInfo{"SAR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
// GROUP 3
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 2), 1, X86InstInfo{"NOT", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 3), 1, X86InstInfo{"NEG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 2), 1, X86InstInfo{"NOT", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 3), 1, X86InstInfo{"NEG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 4), 1, X86InstInfo{"MUL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 5), 1, X86InstInfo{"IMUL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 1, X86InstInfo{"DIV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 7), 1, X86InstInfo{"IDIV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, X86InstInfo{"NOT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, X86InstInfo{"NEG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, X86InstInfo{"NOT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, X86InstInfo{"NEG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 4), 1, X86InstInfo{"MUL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 5), 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 6), 1, X86InstInfo{"DIV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 7), 1, X86InstInfo{"IDIV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
// GROUP 4
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 0), 1, X86InstInfo{"INC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 1), 1, X86InstInfo{"DEC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 0), 1, X86InstInfo{"INC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 1), 1, X86InstInfo{"DEC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 2), 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
// GROUP 5
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 0), 1, X86InstInfo{"INC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 1), 1, X86InstInfo{"DEC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 0), 1, X86InstInfo{"INC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 1), 1, X86InstInfo{"DEC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 2), 1, X86InstInfo{"CALL", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_MODRM | FLAGS_BLOCK_END | FLAGS_CALL , 0}},
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 3), 1, X86InstInfo{"CALLF", TYPE_INST, FLAGS_SETS_RIP | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_BLOCK_END, 0}},
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 4), 1, X86InstInfo{"JMP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_MODRM | FLAGS_BLOCK_END , 0}},
@@ -196,7 +196,7 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 0), 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT, 1}},
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 7), 1, X86InstInfo{"XABORT", TYPE_INST, FLAGS_MODRM, 1}},
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 7), 1, X86InstInfo{"XBEGIN", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT | FLAGS_SETS_RIP | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
};
@@ -139,37 +139,37 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
{OPD(TYPE_GROUP_8, PF_NONE, 1), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_NONE, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_NONE, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_NONE, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_NONE, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_NONE, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_NONE, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_NONE, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_NONE, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_NONE, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_NONE, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_F3, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_F3, 1), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_F3, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_F3, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_F3, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_F3, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_F3, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_F3, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_F3, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_F3, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_F3, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_F3, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_66, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_66, 1), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_66, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_66, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_66, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_66, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_66, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_66, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_66, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_66, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_66, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_66, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_F2, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_F2, 1), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_F2, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_F2, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_8, PF_F2, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_F2, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_F2, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_F2, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
{OPD(TYPE_GROUP_8, PF_F2, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_F2, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_F2, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
{OPD(TYPE_GROUP_8, PF_F2, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
// GROUP 9
@@ -179,7 +179,7 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
// CMPXCHG8B/16B works with all prefixes
// Tooling fails to decode CMPXCHG with prefix
{OPD(TYPE_GROUP_9, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_NONE, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_9, PF_NONE, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
{OPD(TYPE_GROUP_9, PF_NONE, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_NONE, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_NONE, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
@@ -188,7 +188,7 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
{OPD(TYPE_GROUP_9, PF_NONE, 7), 1, X86InstInfo{"RDSEED", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0}},
{OPD(TYPE_GROUP_9, PF_F3, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_F3, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_9, PF_F3, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
{OPD(TYPE_GROUP_9, PF_F3, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_F3, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_F3, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
@@ -197,7 +197,7 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"RDPID", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0}},
{OPD(TYPE_GROUP_9, PF_66, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_66, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_9, PF_66, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
{OPD(TYPE_GROUP_9, PF_66, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_66, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_66, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
@@ -206,7 +206,7 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
{OPD(TYPE_GROUP_9, PF_66, 7), 1, X86InstInfo{"RDSEED", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0}},
{OPD(TYPE_GROUP_9, PF_F2, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_F2, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SUPPORTS_LOCK, 0}},
{OPD(TYPE_GROUP_9, PF_F2, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
{OPD(TYPE_GROUP_9, PF_F2, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_F2, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{OPD(TYPE_GROUP_9, PF_F2, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
@@ -50,7 +50,7 @@ constexpr std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableO
{((3 << 3) | 1), 1, X86InstInfo{"RDTSCP", TYPE_INST, FLAGS_NONE, 0}},
{((3 << 3) | 2), 1, X86InstInfo{"MONITORX", TYPE_PRIV, FLAGS_NONE, 0}},
{((3 << 3) | 3), 1, X86InstInfo{"MWAITX", TYPE_PRIV, FLAGS_NONE, 0}},
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS, 0}},
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_SRC_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
{((3 << 3) | 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{((3 << 3) | 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{((3 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
@@ -26,8 +26,8 @@ enum Secondary_LUT {
constexpr std::array<X86InstInfo[2], ENTRY_MAX> Secondary_ArchSelect_LUT = {{
{
{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 0, { .OpDispatch = &IR::OpDispatchBuilder::NOPOp } },
{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SyscallOp, true> } },
{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, { .OpDispatch = &IR::OpDispatchBuilder::NOPOp } },
{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SyscallOp, true> } },
},
{
{"PUSH FS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX> } },
@@ -61,8 +61,8 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
{0x05, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_05] }}},
{0x06, 1, X86InstInfo{"CLTS", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
{0x07, 1, X86InstInfo{"SYSRET", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
{0x08, 1, X86InstInfo{"INVD", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
{0x09, 1, X86InstInfo{"WBINVD", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0}},
{0x09, 1, X86InstInfo{"WBINVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0}},
{0x0A, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0}},
{0x0C, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
@@ -204,24 +204,24 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
{0xA0, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A0] }}},
{0xA1, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A1] }}},
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0}},
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
{0xA4, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1}},
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0}},
{0xA6, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
{0xA8, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A8] }}},
{0xA9, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A9] }}},
{0xAA, 1, X86InstInfo{"RSM", TYPE_PRIV, FLAGS_NO_OVERLAY, 0}},
{0xAB, 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
{0xAB, 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
{0xAC, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1}},
{0xAD, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
{0xAD, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0}},
{0xAE, 1, X86InstInfo{"", TYPE_GROUP_15, FLAGS_NO_OVERLAY, 0}},
{0xAF, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
{0xB0, 1, X86InstInfo{"CMPXCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
{0xB1, 1, X86InstInfo{"CMPXCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
{0xB0, 1, X86InstInfo{"CMPXCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
{0xB1, 1, X86InstInfo{"CMPXCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
{0xB2, 1, X86InstInfo{"LSS", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
{0xB3, 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
{0xB3, 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
{0xB4, 1, X86InstInfo{"LFS", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
{0xB5, 1, X86InstInfo{"LGS", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
{0xB6, 1, X86InstInfo{"MOVZX", TYPE_INST, GenFlagsSrcSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
@@ -229,14 +229,14 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
{0xB8, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
{0xB9, 1, X86InstInfo{"", TYPE_GROUP_10, FLAGS_NO_OVERLAY, 0}},
{0xBA, 1, X86InstInfo{"", TYPE_GROUP_8, FLAGS_NO_OVERLAY, 0}},
{0xBB, 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
{0xBB, 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
{0xBC, 1, X86InstInfo{"BSF", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY66, 0}},
{0xBD, 1, X86InstInfo{"BSR", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY66, 0}},
{0xBE, 1, X86InstInfo{"MOVSX", TYPE_INST, GenFlagsSrcSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
{0xBF, 1, X86InstInfo{"MOVSX", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
{0xC0, 1, X86InstInfo{"XADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
{0xC1, 1, X86InstInfo{"XADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
{0xC0, 1, X86InstInfo{"XADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
{0xC1, 1, X86InstInfo{"XADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
{0xC2, 1, X86InstInfo{"CMPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
{0xC3, 1, X86InstInfo{"MOVNTI", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST, 0}},
{0xC4, 1, X86InstInfo{"PINSRW", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX | FLAGS_SF_SRC_GPR, 1}},
@@ -303,7 +303,7 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
{0x3E, 1, X86InstInfo{"CALLBACKRET", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
// This was originally used by VIA to jump to its alternative instruction set. Used for OP_THUNK
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, sizeof(IR::SHA256Sum)}},
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
#endif
};
@@ -312,11 +312,6 @@ namespace AVX128 {
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VSSHR>}, // VPSRAVD
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VUSHL>}, // VPSLLV
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, false>},
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, true>},
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, false>},
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, true>},
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i32Bit>},
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i64Bit>},
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i128Bit>},
@@ -479,7 +474,7 @@ namespace AVX256 {
{OPD(1, 0b00, 0x12), 1, &OpDispatchBuilder::VMOVLPOp},
{OPD(1, 0b01, 0x12), 1, &OpDispatchBuilder::VMOVLPOp},
{OPD(1, 0b10, 0x12), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMOVSLDUPOp, true>},
{OPD(1, 0b10, 0x12), 1, &OpDispatchBuilder::VMOVSLDUPOp},
{OPD(1, 0b11, 0x12), 1, &OpDispatchBuilder::VMOVDDUPOp},
{OPD(1, 0b00, 0x13), 1, &OpDispatchBuilder::VMOVLPOp},
{OPD(1, 0b01, 0x13), 1, &OpDispatchBuilder::VMOVLPOp},
@@ -492,7 +487,7 @@ namespace AVX256 {
{OPD(1, 0b00, 0x16), 1, &OpDispatchBuilder::VMOVHPOp},
{OPD(1, 0b01, 0x16), 1, &OpDispatchBuilder::VMOVHPOp},
{OPD(1, 0b10, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMOVSHDUPOp, true>},
{OPD(1, 0b10, 0x16), 1, &OpDispatchBuilder::VMOVSHDUPOp},
{OPD(1, 0b00, 0x17), 1, &OpDispatchBuilder::VMOVHPOp},
{OPD(1, 0b01, 0x17), 1, &OpDispatchBuilder::VMOVHPOp},
@@ -501,36 +496,36 @@ namespace AVX256 {
{OPD(1, 0b00, 0x29), 1, &OpDispatchBuilder::VMOVAPS_VMOVAPDOp},
{OPD(1, 0b01, 0x29), 1, &OpDispatchBuilder::VMOVAPS_VMOVAPDOp},
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertCVTGPR_To_FPR, OpSize::i32Bit>},
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertCVTGPR_To_FPR, OpSize::i64Bit>},
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<OpSize::i32Bit>},
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<OpSize::i64Bit>},
{OPD(1, 0b00, 0x2B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, true>},
{OPD(1, 0b01, 0x2B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, true>},
{OPD(1, 0b00, 0x2B), 1, &OpDispatchBuilder::MOVVectorNTOp},
{OPD(1, 0b01, 0x2B), 1, &OpDispatchBuilder::MOVVectorNTOp},
{OPD(1, 0b10, 0x2C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i32Bit, false>},
{OPD(1, 0b11, 0x2C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i64Bit, false>},
{OPD(1, 0b10, 0x2C), 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, false>},
{OPD(1, 0b11, 0x2C), 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, false>},
{OPD(1, 0b10, 0x2D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i32Bit, true>},
{OPD(1, 0b11, 0x2D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i64Bit, true>},
{OPD(1, 0b10, 0x2D), 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, true>},
{OPD(1, 0b11, 0x2D), 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, true>},
{OPD(1, 0b00, 0x2E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i32Bit>},
{OPD(1, 0b01, 0x2E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i64Bit>},
{OPD(1, 0b00, 0x2F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i32Bit>},
{OPD(1, 0b01, 0x2F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i64Bit>},
{OPD(1, 0b00, 0x2E), 1, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
{OPD(1, 0b01, 0x2E), 1, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
{OPD(1, 0b00, 0x2F), 1, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
{OPD(1, 0b01, 0x2F), 1, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
{OPD(1, 0b00, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i32Bit>},
{OPD(1, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i64Bit>},
{OPD(1, 0b00, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VFSQRT, OpSize::i32Bit>},
{OPD(1, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VFSQRT, OpSize::i64Bit>},
{OPD(1, 0b10, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp, IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp, IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b10, 0x51), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x51), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b00, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
{OPD(1, 0b10, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp, IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b10, 0x52), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b00, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
{OPD(1, 0b10, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp, IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b10, 0x53), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b00, 0x54), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
{OPD(1, 0b01, 0x54), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
@@ -546,42 +541,42 @@ namespace AVX256 {
{OPD(1, 0b00, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
{OPD(1, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFADD, OpSize::i64Bit>},
{OPD(1, 0b10, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b10, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b00, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
{OPD(1, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMUL, OpSize::i64Bit>},
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b00, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit, true>},
{OPD(1, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit, true>},
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit>},
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, false, true>},
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, true, true>},
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, false, true>},
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
{OPD(1, 0b00, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
{OPD(1, 0b01, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
{OPD(1, 0b10, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b10, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b00, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
{OPD(1, 0b01, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMIN, OpSize::i64Bit>},
{OPD(1, 0b10, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b10, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b00, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFDIV, OpSize::i32Bit>},
{OPD(1, 0b01, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFDIV, OpSize::i64Bit>},
{OPD(1, 0b10, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b10, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b00, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
{OPD(1, 0b01, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMAX, OpSize::i64Bit>},
{OPD(1, 0b10, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b10, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
{OPD(1, 0b11, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
{OPD(1, 0b01, 0x60), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPUNPCKLOp, OpSize::i8Bit>},
{OPD(1, 0b01, 0x61), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPUNPCKLOp, OpSize::i16Bit>},
@@ -612,8 +607,8 @@ namespace AVX256 {
{OPD(1, 0b00, 0x77), 1, &OpDispatchBuilder::VZEROOp},
{OPD(1, 0b01, 0x7C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHADDPOp, IR::OP_VFADDP, OpSize::i64Bit>},
{OPD(1, 0b11, 0x7C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHADDPOp, IR::OP_VFADDP, OpSize::i32Bit>},
{OPD(1, 0b01, 0x7C), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VFADDP, OpSize::i64Bit>},
{OPD(1, 0b11, 0x7C), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VFADDP, OpSize::i32Bit>},
{OPD(1, 0b01, 0x7D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHSUBPOp, OpSize::i64Bit>},
{OPD(1, 0b11, 0x7D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHSUBPOp, OpSize::i32Bit>},
@@ -623,19 +618,19 @@ namespace AVX256 {
{OPD(1, 0b01, 0x7F), 1, &OpDispatchBuilder::VMOVAPS_VMOVAPDOp},
{OPD(1, 0b10, 0x7F), 1, &OpDispatchBuilder::VMOVUPS_VMOVUPDOp},
{OPD(1, 0b00, 0xC2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVFCMPOp, OpSize::i32Bit>},
{OPD(1, 0b01, 0xC2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVFCMPOp, OpSize::i64Bit>},
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalarFCMPOp, OpSize::i32Bit>},
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalarFCMPOp, OpSize::i64Bit>},
{OPD(1, 0b00, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<OpSize::i32Bit>},
{OPD(1, 0b01, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<OpSize::i64Bit>},
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i32Bit>},
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i64Bit>},
{OPD(1, 0b01, 0xC4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPINSRBWOp, OpSize::i16Bit>},
{OPD(1, 0b01, 0xC4), 1, &OpDispatchBuilder::VPINSRWOp},
{OPD(1, 0b01, 0xC5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
{OPD(1, 0b00, 0xC6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VSHUFOp, OpSize::i32Bit>},
{OPD(1, 0b01, 0xC6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VSHUFOp, OpSize::i64Bit>},
{OPD(1, 0b01, 0xD0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VADDSUBPOp, OpSize::i64Bit>},
{OPD(1, 0b11, 0xD0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VADDSUBPOp, OpSize::i32Bit>},
{OPD(1, 0b01, 0xD0), 1, &OpDispatchBuilder::VADDSUBPOp<OpSize::i64Bit>},
{OPD(1, 0b11, 0xD0), 1, &OpDispatchBuilder::VADDSUBPOp<OpSize::i32Bit>},
{OPD(1, 0b01, 0xD1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSRLDOp, OpSize::i16Bit>},
{OPD(1, 0b01, 0xD2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSRLDOp, OpSize::i32Bit>},
@@ -658,14 +653,14 @@ namespace AVX256 {
{OPD(1, 0b01, 0xE1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSRAOp, OpSize::i16Bit>},
{OPD(1, 0b01, 0xE2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSRAOp, OpSize::i32Bit>},
{OPD(1, 0b01, 0xE3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
{OPD(1, 0b01, 0xE4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMULHWOp, false>},
{OPD(1, 0b01, 0xE5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMULHWOp, true>},
{OPD(1, 0b01, 0xE4), 1, &OpDispatchBuilder::VPMULHWOp<false>},
{OPD(1, 0b01, 0xE5), 1, &OpDispatchBuilder::VPMULHWOp<true>},
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i64Bit, false, true>},
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, true, true>},
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i64Bit, true, true>},
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
{OPD(1, 0b01, 0xE7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, true>},
{OPD(1, 0b01, 0xE7), 1, &OpDispatchBuilder::MOVVectorNTOp},
{OPD(1, 0b01, 0xE8), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
{OPD(1, 0b01, 0xE9), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
@@ -676,11 +671,11 @@ namespace AVX256 {
{OPD(1, 0b01, 0xEE), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VSMAX, OpSize::i16Bit>},
{OPD(1, 0b01, 0xEF), 1, &OpDispatchBuilder::AVXVectorXOROp},
{OPD(1, 0b11, 0xF0), 1, &OpDispatchBuilder::VMOVUPS_VMOVUPDOp},
{OPD(1, 0b11, 0xF0), 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
{OPD(1, 0b01, 0xF1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSLLOp, OpSize::i16Bit>},
{OPD(1, 0b01, 0xF2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSLLOp, OpSize::i32Bit>},
{OPD(1, 0b01, 0xF3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSLLOp, OpSize::i64Bit>},
{OPD(1, 0b01, 0xF4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMULLOp, OpSize::i32Bit, false>},
{OPD(1, 0b01, 0xF4), 1, &OpDispatchBuilder::VPMULLOp<OpSize::i32Bit, false>},
{OPD(1, 0b01, 0xF5), 1, &OpDispatchBuilder::VPMADDWDOp},
{OPD(1, 0b01, 0xF6), 1, &OpDispatchBuilder::VPSADBWOp},
{OPD(1, 0b01, 0xF7), 1, &OpDispatchBuilder::MASKMOVOp},
@@ -694,8 +689,8 @@ namespace AVX256 {
{OPD(1, 0b01, 0xFE), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
{OPD(2, 0b01, 0x00), 1, &OpDispatchBuilder::VPSHUFBOp},
{OPD(2, 0b01, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHADDPOp, IR::OP_VADDP, OpSize::i16Bit>},
{OPD(2, 0b01, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHADDPOp, IR::OP_VADDP, OpSize::i32Bit>},
{OPD(2, 0b01, 0x01), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VADDP, OpSize::i16Bit>},
{OPD(2, 0b01, 0x02), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VADDP, OpSize::i32Bit>},
{OPD(2, 0b01, 0x03), 1, &OpDispatchBuilder::VPHADDSWOp},
{OPD(2, 0b01, 0x04), 1, &OpDispatchBuilder::VPMADDUBSWOp},
@@ -703,14 +698,14 @@ namespace AVX256 {
{OPD(2, 0b01, 0x06), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPHSUBOp, OpSize::i32Bit>},
{OPD(2, 0b01, 0x07), 1, &OpDispatchBuilder::VPHSUBSWOp},
{OPD(2, 0b01, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSIGN, OpSize::i8Bit>},
{OPD(2, 0b01, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSIGN, OpSize::i16Bit>},
{OPD(2, 0b01, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSIGN, OpSize::i32Bit>},
{OPD(2, 0b01, 0x08), 1, &OpDispatchBuilder::VPSIGN<OpSize::i8Bit>},
{OPD(2, 0b01, 0x09), 1, &OpDispatchBuilder::VPSIGN<OpSize::i16Bit>},
{OPD(2, 0b01, 0x0A), 1, &OpDispatchBuilder::VPSIGN<OpSize::i32Bit>},
{OPD(2, 0b01, 0x0B), 1, &OpDispatchBuilder::VPMULHRSWOp},
{OPD(2, 0b01, 0x0C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPERMILRegOp, OpSize::i32Bit>},
{OPD(2, 0b01, 0x0D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPERMILRegOp, OpSize::i64Bit>},
{OPD(2, 0b01, 0x0E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VTESTPOp, OpSize::i32Bit>},
{OPD(2, 0b01, 0x0F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VTESTPOp, OpSize::i64Bit>},
{OPD(2, 0b01, 0x0C), 1, &OpDispatchBuilder::VPERMILRegOp<OpSize::i32Bit>},
{OPD(2, 0b01, 0x0D), 1, &OpDispatchBuilder::VPERMILRegOp<OpSize::i64Bit>},
{OPD(2, 0b01, 0x0E), 1, &OpDispatchBuilder::VTESTPOp<OpSize::i32Bit>},
{OPD(2, 0b01, 0x0F), 1, &OpDispatchBuilder::VTESTPOp<OpSize::i64Bit>},
{OPD(2, 0b01, 0x13), 1, &OpDispatchBuilder::VCVTPH2PSOp},
{OPD(2, 0b01, 0x16), 1, &OpDispatchBuilder::VPERMDOp},
@@ -722,28 +717,28 @@ namespace AVX256 {
{OPD(2, 0b01, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VABS, OpSize::i16Bit>},
{OPD(2, 0b01, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
{OPD(2, 0b01, 0x20), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, true>},
{OPD(2, 0b01, 0x21), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, true>},
{OPD(2, 0b01, 0x22), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, true>},
{OPD(2, 0b01, 0x23), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, true>},
{OPD(2, 0b01, 0x24), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, true>},
{OPD(2, 0b01, 0x25), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, true>},
{OPD(2, 0b01, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, true>},
{OPD(2, 0b01, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, true>},
{OPD(2, 0b01, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, true>},
{OPD(2, 0b01, 0x23), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, true>},
{OPD(2, 0b01, 0x24), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, true>},
{OPD(2, 0b01, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, true>},
{OPD(2, 0b01, 0x28), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMULLOp, OpSize::i32Bit, true>},
{OPD(2, 0b01, 0x28), 1, &OpDispatchBuilder::VPMULLOp<OpSize::i32Bit, true>},
{OPD(2, 0b01, 0x29), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VCMPEQ, OpSize::i64Bit>},
{OPD(2, 0b01, 0x2A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, true>},
{OPD(2, 0b01, 0x2A), 1, &OpDispatchBuilder::MOVVectorNTOp},
{OPD(2, 0b01, 0x2B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPACKUSOp, OpSize::i32Bit>},
{OPD(2, 0b01, 0x2C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMASKMOVOp, OpSize::i32Bit, false>},
{OPD(2, 0b01, 0x2D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMASKMOVOp, OpSize::i64Bit, false>},
{OPD(2, 0b01, 0x2E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMASKMOVOp, OpSize::i32Bit, true>},
{OPD(2, 0b01, 0x2F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMASKMOVOp, OpSize::i64Bit, true>},
{OPD(2, 0b01, 0x2C), 1, &OpDispatchBuilder::VMASKMOVOp<OpSize::i32Bit, false>},
{OPD(2, 0b01, 0x2D), 1, &OpDispatchBuilder::VMASKMOVOp<OpSize::i64Bit, false>},
{OPD(2, 0b01, 0x2E), 1, &OpDispatchBuilder::VMASKMOVOp<OpSize::i32Bit, true>},
{OPD(2, 0b01, 0x2F), 1, &OpDispatchBuilder::VMASKMOVOp<OpSize::i64Bit, true>},
{OPD(2, 0b01, 0x30), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, false>},
{OPD(2, 0b01, 0x31), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, false>},
{OPD(2, 0b01, 0x32), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, false>},
{OPD(2, 0b01, 0x33), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, false>},
{OPD(2, 0b01, 0x34), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, false>},
{OPD(2, 0b01, 0x35), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, false>},
{OPD(2, 0b01, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, false>},
{OPD(2, 0b01, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, false>},
{OPD(2, 0b01, 0x32), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, false>},
{OPD(2, 0b01, 0x33), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, false>},
{OPD(2, 0b01, 0x34), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, false>},
{OPD(2, 0b01, 0x35), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, false>},
{OPD(2, 0b01, 0x36), 1, &OpDispatchBuilder::VPERMDOp},
{OPD(2, 0b01, 0x37), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VCMPGT, OpSize::i64Bit>},
@@ -757,16 +752,11 @@ namespace AVX256 {
{OPD(2, 0b01, 0x3F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VUMAX, OpSize::i32Bit>},
{OPD(2, 0b01, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VMUL, OpSize::i32Bit>},
{OPD(2, 0b01, 0x41), 1, &OpDispatchBuilder::AVXPHMINPOSUWOp},
{OPD(2, 0b01, 0x41), 1, &OpDispatchBuilder::PHMINPOSUWOp},
{OPD(2, 0b01, 0x45), 1, &OpDispatchBuilder::VPSRLVOp},
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::VPSRAVDOp},
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::VPSLLVOp},
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, false>},
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, true>},
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, false>},
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, true>},
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i32Bit>},
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i64Bit>},
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i128Bit>},
@@ -774,13 +764,13 @@ namespace AVX256 {
{OPD(2, 0b01, 0x78), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i8Bit>},
{OPD(2, 0b01, 0x79), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i16Bit>},
{OPD(2, 0b01, 0x8C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMASKMOVOp, false>},
{OPD(2, 0b01, 0x8E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMASKMOVOp, true>},
{OPD(2, 0b01, 0x8C), 1, &OpDispatchBuilder::VPMASKMOVOp<false>},
{OPD(2, 0b01, 0x8E), 1, &OpDispatchBuilder::VPMASKMOVOp<true>},
{OPD(2, 0b01, 0x90), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPGATHER, OpSize::i32Bit>},
{OPD(2, 0b01, 0x91), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPGATHER, OpSize::i64Bit>},
{OPD(2, 0b01, 0x92), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPGATHER, OpSize::i32Bit>},
{OPD(2, 0b01, 0x93), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPGATHER, OpSize::i64Bit>},
{OPD(2, 0b01, 0x90), 1, &OpDispatchBuilder::VPGATHER<OpSize::i32Bit>},
{OPD(2, 0b01, 0x91), 1, &OpDispatchBuilder::VPGATHER<OpSize::i64Bit>},
{OPD(2, 0b01, 0x92), 1, &OpDispatchBuilder::VPGATHER<OpSize::i32Bit>},
{OPD(2, 0b01, 0x93), 1, &OpDispatchBuilder::VPGATHER<OpSize::i64Bit>},
{OPD(2, 0b01, 0x96), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, true, 1, 3, 2>}, // VFMADDSUB
{OPD(2, 0b01, 0x97), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, false, 1, 3, 2>}, // VFMSUBADD
@@ -818,7 +808,7 @@ namespace AVX256 {
{OPD(2, 0b01, 0xB6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, true, 2, 3, 1>}, // VFMADDSUB
{OPD(2, 0b01, 0xB7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, false, 2, 3, 1>}, // VFMSUBADD
{OPD(2, 0b01, 0xDB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESImcOp, true>},
{OPD(2, 0b01, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
{OPD(2, 0b01, 0xDC), 1, &OpDispatchBuilder::VAESEncOp},
{OPD(2, 0b01, 0xDD), 1, &OpDispatchBuilder::VAESEncLastOp},
{OPD(2, 0b01, 0xDE), 1, &OpDispatchBuilder::VAESDecOp},
@@ -830,10 +820,10 @@ namespace AVX256 {
{OPD(3, 0b01, 0x04), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPERMILImmOp, OpSize::i32Bit>},
{OPD(3, 0b01, 0x05), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPERMILImmOp, OpSize::i64Bit>},
{OPD(3, 0b01, 0x06), 1, &OpDispatchBuilder::VPERM2Op},
{OPD(3, 0b01, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorRound, OpSize::i32Bit>},
{OPD(3, 0b01, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorRound, OpSize::i64Bit>},
{OPD(3, 0b01, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalarRound, OpSize::i32Bit>},
{OPD(3, 0b01, 0x0B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalarRound, OpSize::i64Bit>},
{OPD(3, 0b01, 0x08), 1, &OpDispatchBuilder::AVXVectorRound<OpSize::i32Bit>},
{OPD(3, 0b01, 0x09), 1, &OpDispatchBuilder::AVXVectorRound<OpSize::i64Bit>},
{OPD(3, 0b01, 0x0A), 1, &OpDispatchBuilder::AVXInsertScalarRound<OpSize::i32Bit>},
{OPD(3, 0b01, 0x0B), 1, &OpDispatchBuilder::AVXInsertScalarRound<OpSize::i64Bit>},
{OPD(3, 0b01, 0x0C), 1, &OpDispatchBuilder::VPBLENDDOp},
{OPD(3, 0b01, 0x0D), 1, &OpDispatchBuilder::VBLENDPDOp},
{OPD(3, 0b01, 0x0E), 1, &OpDispatchBuilder::VPBLENDWOp},
@@ -847,15 +837,15 @@ namespace AVX256 {
{OPD(3, 0b01, 0x18), 1, &OpDispatchBuilder::VINSERTOp},
{OPD(3, 0b01, 0x19), 1, &OpDispatchBuilder::VEXTRACT128Op},
{OPD(3, 0b01, 0x1D), 1, &OpDispatchBuilder::VCVTPS2PHOp},
{OPD(3, 0b01, 0x20), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPINSRBWOp, OpSize::i8Bit>},
{OPD(3, 0b01, 0x20), 1, &OpDispatchBuilder::VPINSRBOp},
{OPD(3, 0b01, 0x21), 1, &OpDispatchBuilder::VINSERTPSOp},
{OPD(3, 0b01, 0x22), 1, &OpDispatchBuilder::VPINSRDQOp},
{OPD(3, 0b01, 0x38), 1, &OpDispatchBuilder::VINSERTOp},
{OPD(3, 0b01, 0x39), 1, &OpDispatchBuilder::VEXTRACT128Op},
{OPD(3, 0b01, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VDPPOp, OpSize::i32Bit>},
{OPD(3, 0b01, 0x41), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VDPPOp, OpSize::i64Bit>},
{OPD(3, 0b01, 0x40), 1, &OpDispatchBuilder::VDPPOp<OpSize::i32Bit>},
{OPD(3, 0b01, 0x41), 1, &OpDispatchBuilder::VDPPOp<OpSize::i64Bit>},
{OPD(3, 0b01, 0x42), 1, &OpDispatchBuilder::VMPSADBWOp},
{OPD(3, 0b01, 0x44), 1, &OpDispatchBuilder::VPCLMULQDQOp},
@@ -865,12 +855,12 @@ namespace AVX256 {
{OPD(3, 0b01, 0x4B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorVariableBlend, OpSize::i64Bit>},
{OPD(3, 0b01, 0x4C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorVariableBlend, OpSize::i8Bit>},
{OPD(3, 0b01, 0x60), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPESTRMOp, true>},
{OPD(3, 0b01, 0x61), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPESTRIOp, true>},
{OPD(3, 0b01, 0x62), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRMOp, true>},
{OPD(3, 0b01, 0x63), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRIOp, true>},
{OPD(3, 0b01, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
{OPD(3, 0b01, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
{OPD(3, 0b01, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
{OPD(3, 0b01, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESKeyGenAssist, true>},
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
};
#undef OPD
@@ -1217,11 +1207,6 @@ auto BaseTableLambda = [](const auto RuntimeTable) consteval {
{OPD(2, 0b01, 0x46), 1, X86InstInfo{"VPSRAVD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x47), 1, X86InstInfo{"VPSLLV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x50), 1, X86InstInfo{"VPDPBUSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x51), 1, X86InstInfo{"VPDPBUSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x52), 1, X86InstInfo{"VPDPWSSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x53), 1, X86InstInfo{"VPDPWSSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x58), 1, X86InstInfo{"VPBROADCASTD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x59), 1, X86InstInfo{"VPBROADCASTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x5A), 1, X86InstInfo{"VBROADCASTI128", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_L_1 | FLAGS_SF_MOD_MEM_ONLY | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
@@ -128,7 +128,6 @@ struct DecodedOperand {
RIPRelativeRelocation,
Literal,
LiteralRelocation,
LiteralPatchable,
SIB,
SIBRelocation
};
@@ -160,9 +159,6 @@ struct DecodedOperand {
bool IsLiteralRelocation() const {
return Type == OpType::LiteralRelocation;
}
bool IsLiteralPatchable() const {
return Type == OpType::LiteralPatchable;
}
bool IsSIB() const {
return Type == OpType::SIB;
}
@@ -171,7 +167,7 @@ struct DecodedOperand {
}
uint64_t Literal() const {
LOGMAN_THROW_A_FMT(IsLiteral() || IsLiteralPatchable(), "Precondition: must be a literal");
LOGMAN_THROW_A_FMT(IsLiteral(), "Precondition: must be a literal");
return Data.Literal.Value;
}
@@ -185,14 +181,10 @@ struct DecodedOperand {
struct {
int64_t Displacement;
uint8_t GPR;
bool PatchableDisp;
uint8_t DispOffset;
} GPRIndirect; // Shared with GPRIndirectRelocation
struct {
int64_t Value;
bool PatchableDisp;
uint8_t DispOffset;
} RIPLiteral; // Shared with RIPLiteralRelocation
struct LiteralType {
@@ -204,20 +196,12 @@ struct DecodedOperand {
int64_t EntrypointOffset;
} LiteralRelocation;
struct {
uint64_t Value;
uint8_t Size;
uint8_t FieldOffset;
uint8_t Width;
} LiteralPatchable;
struct {
int64_t Offset;
uint8_t Scale;
uint8_t Index; // ~0 invalid
uint8_t Base; // ~0 invalid
bool PatchableDisp;
uint8_t DispOffset;
} SIB; // Shared with SIBRelocation
} SIB; // Shared with SIBRelocation
};
TypeUnion Data;
@@ -355,8 +339,11 @@ namespace InstFlags {
constexpr InstFlagType FLAGS_X87_FLAGS = (1ULL << 10);
// Non-XMM subflags
constexpr InstFlagType FLAGS_SF_REX_IN_BYTE = (1ULL << 11);
// subflag [15:12] unused
constexpr InstFlagType FLAGS_SF_DST_RAX = (1ULL << 11);
constexpr InstFlagType FLAGS_SF_DST_RDX = (1ULL << 12);
constexpr InstFlagType FLAGS_SF_SRC_RAX = (1ULL << 13);
constexpr InstFlagType FLAGS_SF_SRC_RCX = (1ULL << 14);
constexpr InstFlagType FLAGS_SF_REX_IN_BYTE = (1ULL << 15);
// XMM subflags
constexpr InstFlagType FLAGS_SF_UNUSED = (1ULL << 11); // No assigned behavior yet
@@ -405,12 +392,8 @@ namespace InstFlags {
constexpr InstFlagType FLAGS_REX_W_1 = (1ULL << 29);
constexpr InstFlagType FLAGS_CALL = (1ULL << 30);
constexpr InstFlagType FLAGS_SUPPORTS_LOCK = (1ULL << 31);
constexpr InstFlagType FLAGS_LITERAL_PATCHABLE = (1ULL << 32);
// Flags [57..33]: Undefined
// Flags [60..58]: Dst size
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
// Flags [63..61]: Src size
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
constexpr InstFlagType SIZE_MASK = 0b111;
@@ -423,6 +406,13 @@ namespace InstFlags {
constexpr InstFlagType SIZE_256BIT = 0b110;
constexpr InstFlagType SIZE_64BITDEF = 0b111; // Default mode is 64bit instead of typical 32bit
#ifndef _WIN32
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY;
#else
// Syscall ends a block on WIN32 because the instruction can update the CPU's RIP.
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY | FLAGS_BLOCK_END;
#endif
constexpr InstFlagType GetSizeDstFlags(InstFlagType Flags) {
return (Flags >> FLAGS_SIZE_DST_OFF) & SIZE_MASK;
}
@@ -216,7 +216,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F64OpTable = {{
// 5 = Invalid
{OPDReg(0xDD, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSAVE},
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, false>},
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::X87FNSTSW},
{OPD(0xDD, 0xC0), 8, &OpDispatchBuilder::X87FFREE},
{OPD(0xDD, 0xC8), 8, &OpDispatchBuilder::FXCH},
@@ -284,7 +284,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F64OpTable = {{
{OPD(0xDF, 0xD0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
{OPD(0xDF, 0xD8), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, true>},
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::X87FNSTSW},
{OPD(0xDF, 0xE8), 8,
&OpDispatchBuilder::Bind<&OpDispatchBuilder::FCOMIF64, OpSize::f80Bit, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
{OPD(0xDF, 0xF0), 8,
@@ -483,7 +483,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F80OpTable = {{
// 5 = Invalid
{OPDReg(0xDD, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSAVE},
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, false>},
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::X87FNSTSW},
{OPD(0xDD, 0xC0), 8, &OpDispatchBuilder::X87FFREE},
{OPD(0xDD, 0xC8), 8, &OpDispatchBuilder::FXCH},
@@ -545,7 +545,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F80OpTable = {{
{OPD(0xDF, 0xD0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
{OPD(0xDF, 0xD8), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, true>},
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::X87FNSTSW},
{OPD(0xDF, 0xE8), 8,
&OpDispatchBuilder::Bind<&OpDispatchBuilder::FCOMI, OpSize::f80Bit, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
{OPD(0xDF, 0xF0), 8,
@@ -794,7 +794,7 @@ auto GenerateX87TableLambda = [](const auto DispatchTable) consteval {
// / 3
{OPD(0xDF, 0xD8), 8, X86InstInfo{"FSTP", TYPE_X87, FLAGS_SF_MOD_DST | FLAGS_POP, 0}},
// / 4
{OPD(0xDF, 0xE0), 1, X86InstInfo{"FNSTSW", TYPE_INST, GenFlagsSameSize(SIZE_16BIT), 0}},
{OPD(0xDF, 0xE0), 1, X86InstInfo{"FNSTSW", TYPE_INST, GenFlagsSameSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0}},
{OPD(0xDF, 0xE1), 7, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
// / 5
{OPD(0xDF, 0xE8), 8, X86InstInfo{"FUCOMIP", TYPE_INST, FLAGS_POP, 0}},
+2 -2
View File
@@ -643,7 +643,7 @@ public:
auto IROp = Node.GetNode(BaseList)->Op(IRList);
if (IROp->Op == OP_BEGINBLOCK) {
auto BeginBlock = IROp->C<IROp_BeginBlock>();
auto BeginBlock = IROp->C<IROp_EndBlock>();
Node = BeginBlock->BlockHeader;
} else if (IROp->Op == OP_CODEBLOCK) {
@@ -675,7 +675,7 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
[[nodiscard]]
bool IsBlockExit(FEXCore::IR::IROps Op);
void Dump(fextl::ostringstream* out, const IRListView* IR);
void Dump(fextl::stringstream* out, const IRListView* IR);
constexpr auto format_as(FEXCore::IR::NodeID ID) {
return ID.Value;
+91 -219
View File
@@ -195,13 +195,13 @@
"HasSideEffects": true
},
"GPR = ValidateCode GPR:$crc, GPR:$Address, u8:$CodeLength": {
"GPR = ValidateCode Array16:$CodeOriginal, GPR:$Address, u8:$CodeLength": {
"HasSideEffects": true,
"HasDest": true,
"DestSize": "OpSize::i64Bit"
},
"ThreadRemoveCodeEntry GPR:$EntryToInvalidate, GPR:$NewRIP": {
"ThreadRemoveCodeEntry": {
"HasSideEffects": true
},
@@ -275,8 +275,7 @@
"The boolean argument asks if we should be reading the reseeded number or not",
"Reseeded RNG calculation is more expensive and will be heavier to use",
"Returns the 64-bit number",
"Falls back to a host RNG call when the hardware doesn't support it",
"Sets the Z flag if the number is invalid.",
"Sets the Z flag if the number is valid.",
"RNG hardware is allowed to fail early and return. Software must always check this"
],
"HasSideEffects": true,
@@ -313,8 +312,8 @@
"HasSideEffects": true,
"RAOverride": "2"
},
"ExitFunction OpSize:#Size, GPR:$NewRIP, BranchHint:$Hint, GPR:$CallReturnAddress, SSA:$CallReturnBlock, i64:$PatchSiteAddress{0}, i64:$PatchSiteSize{0}": {
"Desc": ["Exits the current JIT function with a target RIP - optionally patchable from guest bytes for caching"
"ExitFunction OpSize:#Size, GPR:$NewRIP, BranchHint:$Hint, GPR:$CallReturnAddress, SSA:$CallReturnBlock": {
"Desc": ["Exits the current JIT function with a target RIP"
],
"Inline": ["Any"],
"HasSideEffects": true,
@@ -327,10 +326,11 @@
"CallbackReturn": {
"HasSideEffects": true
},
"Syscall": {
"GPR = Syscall GPR:$SyscallID, GPR:$Arg0, GPR:$Arg1, GPR:$Arg2, GPR:$Arg3, GPR:$Arg4, GPR:$Arg5": {
"HasSideEffects": true,
"Desc": ["Dispatches a guest syscall through to the SyscallHandler class"
]
],
"DestSize": "OpSize::i64Bit"
},
"Thunk GPR:$ArgPtr, SHA256Sum:$ThunkNameHash": {
@@ -717,7 +717,7 @@
"HasSideEffects": true
},
"CacheLineClean GPR:$Addr": {
"Desc": ["Does a 64 byte cacheline clean at the address specified",
"Desc": ["Does a 64 byte cacheline cleanat the address specified",
"Only cleans the data cachelines. Doesn't do any zeroing",
"Skips the invalidation step of the CacheLineClear operation"
],
@@ -954,27 +954,6 @@
]
},
"GPR = PatchableGuestData OpSize:#Size, i64:$Value, i64:$SiteAddress, i64:$SiteSize": {
"Desc": ["Loads Value in a patchable way",
"On disk cache load the value is patched from live guest bytes at SiteAddress"
],
"DestSize": "Size"
},
"GPR = PatchableGuestRIP OpSize:#Size, i64:$Value, i64:$SiteAddress, i64:$SiteSize": {
"Desc": ["Loads GuestRIP-relative Value in a patchable way",
"On disk cache load the value is patched from live guest PC and live displacement at SiteAddress"
],
"DestSize": "Size"
},
"GPR = PatchableGuestCRC OpSize:#Size, i64:$Value, i64:$GuestAddress, i64:$GuestSize": {
"Desc": ["Loads Guest CRC in a patchable way",
"On disk cache load the value is recomputed from live guest bytes and patched"
],
"DestSize": "Size"
},
"GPR = Constant i64:$Constant, ConstPad:$Pad{IR::ConstPad::NoPad}, i32:$MaxBytes{0}": {
"Desc": ["Generates a 64bit constant inside of a GPR",
"Unsupported to create a constant in FPR"
@@ -1885,9 +1864,9 @@
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = VNot OpSize:#RegisterSize, FPR:$Vector": {
"FPR = VNot OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i8Bit"
"ElementSize": "ElementSize"
},
"FPR = VAbs OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
@@ -2025,6 +2004,15 @@
"BitShift > 0"
]
},
"FPR = VUShraI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
"TiedSource": 0,
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
]
},
"FPR = VSShrI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
"TiedSource": 0,
"DestSize": "RegisterSize",
@@ -2037,7 +2025,7 @@
"FPR = VUShrNI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
"TiedSource": 0,
"Desc": ["Unsigned shifts right each element and then narrows to the next lower element size"],
"Desc": "Unsigned shifts right each element and then narrows to the next lower element size",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize >> 1",
"EmitValidation": [
@@ -2058,32 +2046,8 @@
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
]
},
"FPR = VRSHRN OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
"TiedSource": 0,
"Desc": ["Rounding shift right each element and then narrows to the next lower element size",
"Writes result to the bottom half of the destination register, upper half is zeroed"
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize >> 1",
"EmitValidation": [
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
]
},
"FPR = VRSHRNPair OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper, u8:$BitShift": {
"Desc": ["Rounding shift right and narrow a pair of vectors into one result"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize >> 1",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit",
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
]
},
"FPR = VSXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": ["Sign extends elements from the source element size to the next size up"],
"Desc": "Sign extends elements from the source element size to the next size up",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
@@ -2095,7 +2059,7 @@
"ElementSize": "ElementSize << 1"
},
"FPR = VSSHLL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift{0}": {
"Desc": ["Sign extends elements from the source element size to the next size up"],
"Desc": "Sign extends elements from the source element size to the next size up",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
@@ -2107,7 +2071,7 @@
"ElementSize": "ElementSize << 1"
},
"FPR = VUXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": ["Zero extends elements from the source element size to the next size up"],
"Desc": "Zero extends elements from the source element size to the next size up",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
@@ -2189,55 +2153,43 @@
"ElementSize": "ElementSize"
},
"FPR = VAnd OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VAndn OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VOrn OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VOr OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VXor OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VXar OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$LHS, FPR:$RHS, u8:$Rotate": {
"Desc": [
"Performs an XOR of corresponding elements and then rotates them right by",
"an amount between [1, ElementSize]"
],
"FPR = VAnd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == IR::OpSize::i256Bit || RegisterSize == IR::OpSize::i128Bit"
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VAndn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VOrn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VOr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VXor OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
@@ -2262,7 +2214,7 @@
},
"FPR = VAddP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
"Desc": ["Does a horizontal pairwise add of elements across the two source vectors"],
"Desc": "Does a horizontal pairwise add of elements across the two source vectors",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
@@ -2319,7 +2271,7 @@
"ElementSize": "ElementSize"
},
"FPR = VFAddP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
"Desc": ["Does a horizontal pairwise add of elements across the two source vectors with float element types"],
"Desc": "Does a horizontal pairwise add of elements across the two source vectors with float element types",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
@@ -2369,64 +2321,29 @@
"ElementSize": "ElementSize << 1"
},
"FPR = VUMull2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Multiplies the high elements with size extension"],
"Desc": "Multiplies the high elements with size extension",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
"FPR = VSMull2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Multiplies the high elements with size extension"],
"Desc": "Multiplies the high elements with size extension",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
"FPR = VUSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Unsigned by signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
"Requires FEAT_I8MM."
],
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i32Bit",
"TiedSource": 0,
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
]
},
"FPR = VSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
"Requires FEAT_DotProd."
],
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i32Bit",
"TiedSource": 0,
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
]
},
"FPR = VSAddLP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": ["Signed add long pairwise. Adds adjacent pairs of elements in to elements of twice the size.",
"ElementSize is the source size"
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
"FPR = VSAdALP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Acc, FPR:$Vector": {
"Desc": ["Signed add and accumulate long pairwise. Adds adjacent pairs of elements in to the elements of Acc, which are twice the size.",
"ElementSize is the source size"
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1",
"TiedSource": 0
},
"FPR = VUMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Wide unsigned multiply returning the high results"],
"Desc": "Wide unsigned multiply returning the high results",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = VSMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Wide signed multiply returning the high results"],
"Desc": "Wide signed multiply returning the high results",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = VUABDL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Unsigned Absolute Difference Long"],
"Desc": ["Unsigned Absolute Difference Long"
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
@@ -2518,14 +2435,6 @@
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = VUCMPGT OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Vector compare unsigned greater than",
"Each element is compared, if the result is true then the resulting element is ~0, else zero"
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = VFCMPEQ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
@@ -2593,8 +2502,7 @@
},
"GPR = VPCMPESTRX FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u16:$Control": {
"Desc": ["NOTE: Currently unused. The OpcodeDispatcher implements the SSE4.2 string instructions inline.",
"Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
"This will return the intermediate result of a PCMPESTR-type operation, but NOT the final",
"result. This must be derived from the intermediate result",
@@ -2606,8 +2514,7 @@
"JITDispatch": false
},
"GPR = VPCMPISTRX FPR:$LHS, FPR:$RHS, u8:$Control": {
"Desc": ["NOTE: Currently unused. The OpcodeDispatcher implements the SSE4.2 string instructions inline.",
"Performs intermediate behavior analogous to the x86 PCMPISTRI/PCMPISTRM instruction",
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPISTRI/PCMPISTRM instruction",
"This will return the intermediate result of a PCMPISTR-type operation, but NOT the final",
"result. This must be derived from the intermediate result",
@@ -2657,24 +2564,6 @@
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"TiedSource": 2
},
"FPR = VBlendImm OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$LHS, FPR:$RHS, u16:$Selector": {
"Desc": [
"Functions the same way an immediate blend operation on x86 would.",
"That is: (e.g. using 16-bit elements)",
" if (Selector[0] == 1)",
" Dst[15:0] = RHS[15:0]",
" else",
" Dst[15:0] = LHS[15:0]",
" <etc for the rest of the elements along the vector>",
"",
"Note that like x86, due to the selector size, the operation of this IR op",
"uses a 128-bit lane granularity, so each blending selector independently operates",
"on each 128-bit element that composes the vector."
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"TiedSource": 0
}
},
"Conv": {
@@ -2712,7 +2601,7 @@
},
"FPR = Vector_SToF OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": ["Vector op: Converts signed integer to same size float"],
"Desc": "Vector op: Converts signed integer to same size float",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
@@ -2724,12 +2613,12 @@
"ElementSize": "ElementSize"
},
"FPR = Vector_FToZS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": ["Vector op: Converts float to signed integer, rounding towards zero"],
"Desc": "Vector op: Converts float to signed integer, rounding towards zero",
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = Vector_FToF OpSize:#RegisterSize, OpSize:#DestElementSize, FPR:$Vector, OpSize:$SrcElementSize": {
"Desc": ["Vector op: Converts float from source element size to destination size (fp32<->fp64)"],
"Desc": "Vector op: Converts float from source element size to destination size (fp32<->fp64)",
"DestSize": "RegisterSize",
"ElementSize": "DestElementSize"
},
@@ -2784,74 +2673,75 @@
},
"Crypto": {
"FPR = VAESImc FPR:$Vector": {
"Desc": ["Does a stage of the inverse mix column transformation"],
"Desc": "Does a stage of the inverse mix column transformation",
"DestSize": "OpSize::i128Bit"
},
"FPR = VAESEnc OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
"Desc": ["Does a step of AES encryption"],
"Desc": "Does a step of AES encryption",
"DestSize": "RegisterSize"
},
"FPR = VAESEncLast OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
"Desc": ["Does the last step of AES encryption"],
"Desc": "Does the last step of AES encryption",
"DestSize": "RegisterSize"
},
"FPR = VAESDec OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
"Desc": ["Does a step of AES decryption"],
"Desc": "Does a step of AES decryption",
"DestSize": "RegisterSize"
},
"FPR = VAESDecLast OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
"Desc": ["Does the last step of AES decryption"],
"Desc": "Does the last step of AES decryption",
"DestSize": "RegisterSize"
},
"FPR = VAESKeyGenAssist FPR:$Src, FPR:$KeyGenTBLSwizzle, FPR:$ZeroReg, u8:$RCON": {
"Desc": ["Assists in key generation"],
"Desc": "Assists in key generation",
"DestSize": "OpSize::i128Bit"
},
"FPR = VSha1H FPR:$Src": {
"Desc": ["Does vector scalar SHA1H instruction"],
"Desc": "Does vector scalar SHA1H instruction",
"DestSize": "FEXCore::IR::OpSize::i32Bit"
},
"FPR = VSha1C FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": ["Does vector SHA1C instruction"],
"Desc": "Does vector SHA1C instruction",
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha1M FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": ["Does vector SHA1M instruction"],
"Desc": "Does vector SHA1M instruction",
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha1P FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": ["Does vector SHA1P instruction"],
"Desc": "Does vector SHA1P instruction",
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha1SU1 FPR:$Src1, FPR:$Src2": {
"Desc": ["Does vector scalar SHA1H instruction"],
"Desc": "Does vector scalar SHA1H instruction",
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha256U0 FPR:$Src1, FPR:$Src2": {
"Desc": ["Does vector scalar VSha256U0 instruction"],
"Desc": "Does vector scalar VSha256U0 instruction",
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha256U1 FPR:$Src1, FPR:$Src2": {
"Desc": ["Does vector scalar VSha256U1 instruction"],
"Desc": "Does vector scalar VSha256U1 instruction",
"DestSize": "FEXCore::IR::OpSize::i128Bit"
},
"FPR = VSha256H FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": ["Does vector scalar VSha256H instruction"],
"Desc": "Does vector scalar VSha256H instruction",
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha256H2 FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": ["Does vector scalar VSha256H2 instruction"],
"Desc": "Does vector scalar VSha256H2 instruction",
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"GPR = CRC32 GPR:$Src1, GPR:$Src2, OpSize:$SrcSize": {
"Desc": ["CRC32 using polynomial 0x1EDC6F41"],
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
],
"DestSize": "OpSize::i32Bit"
},
"FPR = PCLMUL OpSize:#RegisterSize, FPR:$Src1, FPR:$Src2, u8:$Selector": {
@@ -2872,11 +2762,11 @@
},
"FPR = F64FPREM FPR:$Src1, FPR:$Src2": {
"DestSize": "OpSize::i64Bit",
"JITDispatch": true
"JITDispatch": false
},
"FPR = F64FPREM1 FPR:$Src1, FPR:$Src2": {
"DestSize": "OpSize::i64Bit",
"JITDispatch": true
"JITDispatch": false
},
"FPR = F64SCALE FPR:$Src1, FPR:$Src2": {
"DestSize": "OpSize::i64Bit",
@@ -2890,10 +2780,6 @@
"DestSize": "OpSize::i64Bit",
"JITDispatch": true
},
"FPR = F64FYL2XP1 FPR:$Src, FPR:$Src2": {
"DestSize": "OpSize::i64Bit",
"JITDispatch": true
},
"FPR = F64TAN FPR:$Src": {
"DestSize": "OpSize::i64Bit",
"JITDispatch": true
@@ -3322,20 +3208,6 @@
"DestSize": "OpSize::i128Bit",
"JITDispatch": false
},
"FPR = F80FYL2XP1Stack": {
"Desc": [
"Computes ST1 * log2(1 + ST0)",
"Stores the result in ST1, and pops the top of the stack.",
"Returns the new value at the top of the stack, i.e. the result of the operation."
],
"HasSideEffects": true,
"DestSize": "OpSize::i128Bit",
"X87": true
},
"FPR = F80FYL2XP1 FPR:$X80Src1, FPR:$X80Src2": {
"DestSize": "OpSize::i128Bit",
"JITDispatch": false
},
"F80VBSLStack OpSize:#RegisterSize, FPR:$VectorMask, u8:$SrcStack1, u8:$SrcStack2": {
"Desc": [
"Does a vector bitwise select.",
+22 -24
View File
@@ -30,19 +30,19 @@ namespace FEXCore::IR {
#include <FEXCore/IR/IRDefines.inc>
static void PrintArg(fextl::ostringstream* out, const IRListView*, const SHA256Sum& Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, const SHA256Sum& Arg) {
*out << fextl::fmt::format("sha256:{:02x}", fmt::join(Arg.data, ""));
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, uint64_t Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, uint64_t Arg) {
*out << fextl::fmt::format("#{:#x}", Arg);
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, const char* const Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, const char* const Arg) {
*out << fextl::fmt::format("'{}'", Arg);
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, CondClass Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg) {
if (Arg == CondClass::AL) {
*out << "ALWAYS";
return;
@@ -55,7 +55,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, CondClass Arg
*out << CondNames[FEXCore::ToUnderlying(Arg)];
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, MemOffsetType Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType Arg) {
static constexpr std::array<std::string_view, 3> Names = {
"SXTX",
"UXTW",
@@ -65,7 +65,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, MemOffsetType
*out << Names[FEXCore::ToUnderlying(Arg)];
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, RegClass Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, RegClass Arg) {
*out << [Arg] {
switch (Arg) {
case RegClass::Invalid: return "Invalid";
@@ -79,7 +79,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, RegClass Arg)
}();
}
static void PrintArg(fextl::ostringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
if (Arg.IsImmediate()) {
auto PhyReg = PhysicalRegister(Arg);
@@ -128,7 +128,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView* IR, OrderedNod
}
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, FenceType Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, FenceType Arg) {
*out << [Arg] {
switch (Arg) {
case FenceType::Load: return "Loads";
@@ -140,7 +140,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, FenceType Arg
}();
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, RoundMode Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, RoundMode Arg) {
*out << [Arg] {
switch (Arg) {
case RoundMode::Nearest: return "Nearest";
@@ -153,7 +153,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, RoundMode Arg
}();
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, ConstPad Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, ConstPad Arg) {
*out << [Arg] {
switch (Arg) {
case ConstPad::NoPad: return "NoPad";
@@ -164,7 +164,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, ConstPad Arg)
}();
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, NamedVectorConstant Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorConstant Arg) {
*out << [Arg] {
// clang-format off
switch (Arg) {
@@ -172,8 +172,6 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, NamedVectorCo
return "u16_incremental_index";
case NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER:
return "u16_incremental_index_upper";
case NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U8_INDEX:
return "u8_incremental_index";
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT:
return "addsubps_invert";
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT_UPPER:
@@ -210,10 +208,6 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, NamedVectorCo
return "movmaskb";
case NamedVectorConstant::NAMED_VECTOR_MOVMASKB_UPPER:
return "movmaskb_upper";
case NamedVectorConstant::NAMED_VECTOR_256_MID_ELEMENT_SWAP:
return "v256_mid_element_swap";
case NamedVectorConstant::NAMED_VECTOR_256_MID_ELEMENT_SWAP_UPPER:
return "v256_mid_element_swap_upper";
case NamedVectorConstant::NAMED_VECTOR_ZERO:
return "vectorzero";
case NamedVectorConstant::NAMED_VECTOR_X87_ONE:
@@ -262,7 +256,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, NamedVectorCo
}();
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, IndexNamedVectorConstant Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, IndexNamedVectorConstant Arg) {
*out << [Arg] {
// clang-format off
switch (Arg) {
@@ -288,7 +282,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, IndexNamedVec
}();
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, OpSize Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, OpSize Arg) {
*out << [Arg] {
switch (Arg) {
case OpSize::iUnsized: return "Unsized";
@@ -305,7 +299,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, OpSize Arg) {
}();
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, FloatCompareOp Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, FloatCompareOp Arg) {
*out << [Arg] {
switch (Arg) {
case FloatCompareOp::EQ: return "FEQ";
@@ -319,14 +313,14 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, FloatCompareO
}();
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, FEXCore::IR::BreakDefinition Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::BreakDefinition Arg) {
*out << "{" << Arg.ErrorRegister << ".";
*out << static_cast<uint32_t>(Arg.Signal) << ".";
*out << static_cast<uint32_t>(Arg.TrapNumber) << ".";
*out << static_cast<uint32_t>(Arg.si_code) << "}";
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, ShiftType Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, ShiftType Arg) {
*out << [Arg] {
switch (Arg) {
case ShiftType::LSL: return "LSL";
@@ -338,7 +332,7 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, ShiftType Arg
}();
}
static void PrintArg(fextl::ostringstream* out, const IRListView*, BranchHint Arg) {
static void PrintArg(fextl::stringstream* out, const IRListView*, BranchHint Arg) {
*out << [Arg] {
switch (Arg) {
case BranchHint::None: return "None";
@@ -350,7 +344,11 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, BranchHint Ar
}();
}
void Dump(fextl::ostringstream* out, const IRListView* IR) {
static void PrintArg(fextl::stringstream* out, const IRListView*, const std::array<uint8_t, 0x10>& Arg) {
*out << fextl::fmt::format("{:02x}", fmt::join(Arg, ""));
}
void Dump(fextl::stringstream* out, const IRListView* IR) {
auto HeaderOp = IR->GetHeader();
int8_t CurrentIndent = 0;
+4 -5
View File
@@ -19,8 +19,6 @@ namespace FEXCore::IR {
static bool IsFragmentExit(FEXCore::IR::IROps Op) {
switch (Op) {
case OP_THREADREMOVECODEENTRY:
case OP_SYSCALL:
case OP_EXITFUNCTION:
case OP_BREAK: return true;
default: return false;
@@ -35,7 +33,7 @@ bool IsBlockExit(FEXCore::IR::IROps Op) {
}
}
RegClass IREmitter::WalkFindRegClass(Ref Node) const {
RegClass IREmitter::WalkFindRegClass(Ref Node) {
auto Class = GetOpRegClass(Node);
switch (Class) {
case RegClass::GPR:
@@ -47,8 +45,9 @@ RegClass IREmitter::WalkFindRegClass(Ref Node) const {
}
// Complex case, needs to be handled on an op by op basis
const uintptr_t DataBegin = DualListData.DataBegin();
const auto* IROp = Node->Op(DataBegin);
uintptr_t DataBegin = DualListData.DataBegin();
FEXCore::IR::IROp_Header* IROp = Node->Op(DataBegin);
switch (IROp->Op) {
case IROps::OP_LOADREGISTER: {
+9 -13
View File
@@ -36,10 +36,6 @@ public:
DualListData.DelayedDisownBuffer();
}
void ValidateDisownedOrFree() const {
DualListData.ValidateDisownedOrFree();
}
IRListView ViewIR() {
return IRListView(&DualListData);
}
@@ -49,7 +45,7 @@ public:
*
* @{ */
RegClass WalkFindRegClass(Ref Node) const;
RegClass WalkFindRegClass(Ref Node);
// These inlining helpers are used by IRDefines.inc so define first.
Ref InlineMem(OpSize Size, Ref Offset, MemOffsetType OffsetType, uint8_t& OffsetScale, bool TSO = false) {
@@ -314,14 +310,14 @@ public:
}
/** @} */
RegClass WalkFindRegClass(OrderedNodeWrapper ssa) const {
auto RealNode = ssa.GetNode(DualListData.ListBegin());
RegClass WalkFindRegClass(OrderedNodeWrapper ssa) {
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
return WalkFindRegClass(RealNode);
}
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t* Constant = nullptr) const {
auto RealNode = ssa.GetNode(DualListData.ListBegin());
const auto* IROp = RealNode->Op(DualListData.DataBegin());
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t* Constant = nullptr) {
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
if (IROp->Op == OP_CONSTANT) {
auto Op = IROp->C<IR::IROp_Constant>();
if (Constant) {
@@ -332,9 +328,9 @@ public:
return false;
}
bool IsValueInlineConstant(OrderedNodeWrapper ssa) const {
auto RealNode = ssa.GetNode(DualListData.ListBegin());
const auto* IROp = RealNode->Op(DualListData.DataBegin());
bool IsValueInlineConstant(OrderedNodeWrapper ssa) {
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
if (IROp->Op == OP_INLINECONSTANT) {
return true;
}
@@ -129,10 +129,6 @@ public:
PoolObject.DelayedDisownBuffer();
}
void ValidateDisownedOrFree() const {
PoolObject.ValidateDisownedOrFree();
}
private:
Utils::PoolBufferWithTimedRetirement<uintptr_t, 5000, 500> PoolObject;
};
@@ -190,12 +186,12 @@ public:
}
[[nodiscard]]
bool PostRA() const {
unsigned PostRA() const {
return GetHeader()->PostRA;
}
[[nodiscard]]
uint32_t SpillSlots() const {
unsigned SpillSlots() const {
return GetHeader()->SpillSlots;
}
+5 -34
View File
@@ -13,7 +13,6 @@ $end_info$
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
namespace FEXCore::IR {
@@ -67,42 +66,25 @@ void PassManager::Finalize() {
}
}
void PassManager::AddDefaultPasses(Context::ContextImpl* ctx) {
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
FEX_CONFIG_OPT(DisablePasses, O0);
// We only specifically disable optimization passes if desired, as IR output should
// still be well-formed regardless of the modifications made to it.
if (!DisablePasses()) {
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures, ctx->Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit));
InsertPass(CreateDeadFlagCalculationEliminination());
}
}
InsertPass(IR::CreateRegisterAllocationPass(&ctx->CPUID), "RA");
void PassManager::AddDefaultValidationPasses() {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
InsertValidationPass(Validation::CreateIRValidation(), "IRValidation");
#endif
}
Pass* PassManager::InsertPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name) {
auto* PassPtr = InsertAt(Passes.end(), std::move(Pass))->get();
AttemptNameMapping(Name, PassPtr);
return PassPtr;
void PassManager::InsertRegisterAllocationPass(FEXCore::Context::ContextImpl* ctx) {
InsertPass(IR::CreateRegisterAllocationPass(&ctx->CPUID), "RA");
}
PassManager::PassArrayType::iterator PassManager::InsertAt(PassArrayType::iterator pos, fextl::unique_ptr<Pass> Pass) {
Pass->RegisterPassManager(this);
return Passes.insert(pos, std::move(Pass));
}
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
void PassManager::InsertValidationPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name) {
Pass->RegisterPassManager(this);
auto* PassPtr = ValidationPasses.emplace_back(std::move(Pass)).get();
AttemptNameMapping(Name, PassPtr);
}
#endif
void PassManager::Run(IREmitter* IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::Run");
@@ -116,15 +98,4 @@ void PassManager::Run(IREmitter* IREmit) {
}
#endif
}
void PassManager::AttemptNameMapping(const fextl::string& Name, Pass* NewPass) {
if (Name.empty()) {
// Empty name is a 'don't care' case. e.g. Passes that just need to run,
// but don't need to be actively looked up.
return;
}
const auto Result = NameToPassMaping.emplace(Name, NewPass);
LOGMAN_THROW_A_FMT(Result.second, "Tried to insert pass with name '{}'. But name is already used", Name);
}
} // namespace FEXCore::IR
+43 -31
View File
@@ -8,18 +8,23 @@ $end_info$
#pragma once
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/ThreadPoolAllocator.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <concepts>
#include <functional>
#include <utility>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::HLE {
class SyscallHandler;
}
namespace FEXCore::IR {
class PassManager;
class IREmitter;
@@ -39,56 +44,63 @@ protected:
class PassManager final {
public:
explicit PassManager(Context::ContextImpl* CTX) {
AddDefaultPasses(CTX);
void AddDefaultPasses(FEXCore::Context::ContextImpl* ctx);
void AddDefaultValidationPasses();
Pass* InsertPass(fextl::unique_ptr<Pass> Pass, fextl::string Name = "") {
auto PassPtr = InsertAt(Passes.end(), std::move(Pass))->get();
if (!Name.empty()) {
NameToPassMaping[Name] = PassPtr;
}
return PassPtr;
}
// Executes all of the passes added to the manager.
// If assertions are enabled, this will also run all validation passes.
void InsertRegisterAllocationPass(FEXCore::Context::ContextImpl* ctx);
void Run(IREmitter* IREmit);
// Inserts a new pass into the manager, optionally also assigning a name to it
// for use in the lookup functions,
Pass* InsertPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name = "");
// Whether or not a pass with the given name is within the manager.
bool HasPass(const fextl::string& Name) const {
bool HasPass(fextl::string Name) const {
return NameToPassMaping.contains(Name);
}
// Retrieves a pass from the manager that has the given name assigned to it.
// Will return nullptr if the pass doesn't exist.
template<std::derived_from<Pass> T>
T* GetPass(const fextl::string& Name) {
return dynamic_cast<T*>(GetPass(Name));
template<typename T>
T* GetPass(fextl::string Name) {
return dynamic_cast<T*>(NameToPassMaping[Name]);
}
Pass* GetPass(const fextl::string& Name) {
const auto Iter = NameToPassMaping.find(Name);
if (Iter == NameToPassMaping.end()) {
return nullptr;
}
return Iter->second;
Pass* GetPass(fextl::string Name) {
return NameToPassMaping[Name];
}
void RegisterSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) {
SyscallHandler = Handler;
}
// Finalizes the pass manager state and assumes no other passes will be added after called.
// This will reorganize the execution order of the passes if necessary.
void Finalize();
protected:
FEXCore::HLE::SyscallHandler* SyscallHandler {};
private:
void AddDefaultPasses(Context::ContextImpl* ctx);
using PassArrayType = fextl::vector<fextl::unique_ptr<Pass>>;
PassArrayType::iterator InsertAt(PassArrayType::iterator pos, fextl::unique_ptr<Pass> Pass);
PassArrayType::iterator InsertAt(PassArrayType::iterator pos, fextl::unique_ptr<Pass> Pass) {
Pass->RegisterPassManager(this);
return Passes.insert(pos, std::move(Pass));
}
PassArrayType Passes;
fextl::unordered_map<fextl::string, Pass*> NameToPassMaping;
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
fextl::vector<fextl::unique_ptr<Pass>> ValidationPasses;
void InsertValidationPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name = "");
#endif
void InsertValidationPass(fextl::unique_ptr<Pass> Pass, fextl::string Name = "") {
Pass->RegisterPassManager(this);
auto PassPtr = ValidationPasses.emplace_back(std::move(Pass)).get();
void AttemptNameMapping(const fextl::string& Name, Pass* NewPass);
if (!Name.empty()) {
NameToPassMaping[Name] = PassPtr;
}
}
#endif
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(PassManagerDumpIR, PASSMANAGERDUMPIR);
+10 -5
View File
@@ -8,18 +8,23 @@ class CPUIDEmu;
struct HostFeatures;
} // namespace FEXCore
namespace FEXCore::Utils {
class IntrusivePooledAllocator;
}
namespace FEXCore::IR {
class Pass;
class RegisterAllocationPass;
fextl::unique_ptr<Pass> CreateDeadFlagCalculationEliminination();
fextl::unique_ptr<Pass> CreateRegisterAllocationPass(const CPUIDEmu* CPUID);
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const HostFeatures&, OpSize GPROpSize);
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(const FEXCore::CPUIDEmu* CPUID);
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&, OpSize GPROpSize);
namespace Validation {
fextl::unique_ptr<Pass> CreateIRValidation();
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
} // namespace Validation
namespace Debug {
fextl::unique_ptr<Pass> CreateIRDumper();
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRDumper();
}
} // namespace FEXCore::IR
@@ -57,7 +57,7 @@ void IRDumper::Run(IREmitter* IREmit) {
}
if (FD.IsValid() || DumpToLog) {
fextl::ostringstream out;
fextl::stringstream out;
FEXCore::IR::Dump(&out, &IR);
if (FD.IsValid()) {
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", +HeaderOp->OriginalRIP, out.str());
@@ -67,7 +67,7 @@ void IRDumper::Run(IREmitter* IREmit) {
}
}
fextl::unique_ptr<Pass> CreateIRDumper() {
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRDumper() {
return fextl::make_unique<IRDumper>();
}
} // namespace FEXCore::IR::Debug
@@ -10,7 +10,6 @@ $end_info$
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/Passes/IRValidation.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
@@ -47,7 +46,7 @@ void IRValidation::Run(IREmitter* IREmit) {
OffsetToBlockMap.clear();
EntryBlock = nullptr;
const auto Count = CurrentIR.GetSSACount();
uint32_t Count = CurrentIR.GetSSACount();
if (Count > MaxNodes) {
NodeIsLive.Realloc(Count);
}
@@ -60,7 +59,7 @@ void IRValidation::Run(IREmitter* IREmit) {
#endif
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
auto BlockIROp = BlockHeader->C<FEXCore::IR::IROp_CodeBlock>();
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
if (!EntryBlock) {
@@ -78,15 +77,15 @@ void IRValidation::Run(IREmitter* IREmit) {
const auto OpSize = IROp->Size;
if (GetHasDest(IROp->Op)) {
// Does the op have an unsized destination?
HadError |= OpSize == IR::OpSize::iInvalid;
// Does the op have a destination of size 0?
if (OpSize == IR::OpSize::iInvalid) {
HadError = true;
Errors << "%" << ID << ": Had destination but with no size" << std::endl;
}
// Does the node have zero uses? Should have been DCE'd
if (CodeNode->GetUses() == 0) {
HadWarning = true;
HadWarning |= true;
Warnings << "%" << ID << ": Destination created but had no uses" << std::endl;
}
@@ -99,26 +98,27 @@ void IRValidation::Run(IREmitter* IREmit) {
// If no register class was assigned
if (AssignedClass == IR::RegClass::Invalid) {
HadError = true;
HadError |= true;
Errors << "%" << ID << ": Had destination but with no register class assigned" << std::endl;
}
// If no physical register was assigned
if (PhyReg.IsInvalid()) {
HadError = true;
HadError |= true;
Errors << "%" << ID << ": Had destination but with no register assigned" << std::endl;
}
// Assigned class wasn't the expected class and it is a non-complex op
if (AssignedClass != ExpectedClass && ExpectedClass != IR::RegClass::Complex) {
HadWarning = true;
HadWarning |= true;
Warnings << "%" << ID << ": Destination had register class " << uint32_t(AssignedClass) << " When register class "
<< uint32_t(ExpectedClass) << " Was expected" << std::endl;
}
}
}
const uint8_t NumArgs = IR::GetRAArgs(IROp->Op);
uint8_t NumArgs = IR::GetRAArgs(IROp->Op);
for (uint32_t i = 0; i < NumArgs; ++i) {
OrderedNodeWrapper Arg = IROp->Args[i];
const auto ArgID = Arg.ID();
@@ -126,6 +126,8 @@ void IRValidation::Run(IREmitter* IREmit) {
continue;
}
IROps Op = CurrentIR.GetOp<IROp_Header>(Arg)->Op;
if (ArgID.IsValid()) {
Uses[ArgID.Value]++;
}
@@ -133,11 +135,10 @@ void IRValidation::Run(IREmitter* IREmit) {
// We do not validate the location of inline constants because it's
// irrelevant, they're ignored by RA and always inlined to where they
// need to be. This lets us pool inline constants globally.
const IROps Op = CurrentIR.GetOp<IROp_Header>(Arg)->Op;
const bool Ignore = (Op == OP_IRHEADER || Op == OP_INLINECONSTANT);
bool Ignore = (Op == OP_IRHEADER || Op == OP_INLINECONSTANT);
if (!Ignore && ArgID.IsValid() && !NodeIsLive.Get(ArgID.Value)) {
HadError = true;
HadError |= true;
Errors << "%" << ID << ": Arg[" << i << "] references invalid %" << ArgID << std::endl;
}
}
@@ -146,6 +147,7 @@ void IRValidation::Run(IREmitter* IREmit) {
switch (IROp->Op) {
case IR::OP_EXITFUNCTION: {
CurrentBlock->HasExit = true;
break;
}
case IR::OP_CONDJUMP: {
@@ -161,7 +163,7 @@ void IRValidation::Run(IREmitter* IREmit) {
const FEXCore::IR::IROp_Header* FalseTargetOp = CurrentIR.GetOp<IROp_Header>(FalseTargetNode);
if (TrueTargetOp->Op != OP_CODEBLOCK) {
HadError = true;
HadError |= true;
Errors << "CondJump %" << ID << ": True Target Jumps to Op that isn't the begining of a block" << std::endl;
} else {
auto Block = OffsetToBlockMap.try_emplace(Op->TrueBlock.ID()).first;
@@ -169,7 +171,7 @@ void IRValidation::Run(IREmitter* IREmit) {
}
if (FalseTargetOp->Op != OP_CODEBLOCK) {
HadError = true;
HadError |= true;
Errors << "CondJump %" << ID << ": False Target Jumps to Op that isn't the begining of a block" << std::endl;
} else {
auto Block = OffsetToBlockMap.try_emplace(Op->FalseBlock.ID()).first;
@@ -185,7 +187,7 @@ void IRValidation::Run(IREmitter* IREmit) {
const FEXCore::IR::IROp_Header* TargetOp = CurrentIR.GetOp<IROp_Header>(TargetNode);
if (TargetOp->Op != OP_CODEBLOCK) {
HadError = true;
HadError |= true;
Errors << "Jump %" << ID << ": Jump to Op that isn't the begining of a block" << std::endl;
} else {
auto Block = OffsetToBlockMap.try_emplace(Op->Header.Args[0].ID()).first;
@@ -202,7 +204,7 @@ void IRValidation::Run(IREmitter* IREmit) {
// Blocks can only have zero (Exit), 1 (Unconditional branch) or 2 (Conditional) successors
size_t NumSuccessors = CurrentBlock->Successors.size();
if (NumSuccessors > 2) {
HadError = true;
HadError |= true;
Errors << "%" << BlockID << " Has " << NumSuccessors << " successors which is too many" << std::endl;
}
@@ -218,7 +220,7 @@ void IRValidation::Run(IREmitter* IREmit) {
{
auto Op = GetOp(CodeCurrent);
if (Op != IR::OP_ENDBLOCK) {
HadError = true;
HadError |= true;
Errors << "%" << BlockID << " Failed to end block with EndBlock" << std::endl;
}
}
@@ -229,7 +231,7 @@ void IRValidation::Run(IREmitter* IREmit) {
{
auto Op = GetOp(CodeCurrent);
if (!IsBlockExit(Op)) {
HadError = true;
HadError |= true;
Errors << "%" << BlockID << " Didn't have a block exit IR op as its last instruction" << std::endl;
}
}
@@ -241,7 +243,7 @@ void IRValidation::Run(IREmitter* IREmit) {
for (uint32_t i = 0; i < CurrentIR.GetSSACount(); i++) {
auto [Node, IROp] = CurrentIR.at(IR::NodeID {i})();
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
HadError = true;
HadError |= true;
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
}
}
@@ -249,7 +251,7 @@ void IRValidation::Run(IREmitter* IREmit) {
HadWarning = false;
if (HadError || HadWarning) {
fextl::ostringstream Out;
fextl::stringstream Out;
FEXCore::IR::Dump(&Out, &CurrentIR);
if (HadError) {
@@ -269,7 +271,7 @@ void IRValidation::Run(IREmitter* IREmit) {
}
}
fextl::unique_ptr<Pass> CreateIRValidation() {
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation() {
return fextl::make_unique<IRValidation>();
}
} // namespace FEXCore::IR::Validation
@@ -8,16 +8,20 @@
namespace FEXCore::IR::Validation {
struct BlockInfo {
bool HasExit;
const OrderedNode* BlockNode;
fextl::vector<OrderedNode*> Predecessors;
fextl::vector<OrderedNode*> Successors;
};
class IRValidation final : public FEXCore::IR::Pass {
public:
~IRValidation();
void Run(IREmitter* IREmit) override;
private:
struct BlockInfo {
fextl::vector<OrderedNode*> Predecessors;
fextl::vector<OrderedNode*> Successors;
};
BitSet<uint64_t> NodeIsLive {};
OrderedNode* EntryBlock {};
@@ -7,7 +7,6 @@ $end_info$
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/X86Enums.h>
@@ -197,7 +196,7 @@ unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClass Cond)
}
}
static constexpr FlagInfo ClassifyConst(IROps Op) {
constexpr FlagInfo ClassifyConst(IROps Op) {
switch (Op) {
case OP_ANDWITHFLAGS:
return FlagInfo::Pack({
@@ -333,15 +332,15 @@ static constexpr FlagInfo ClassifyConst(IROps Op) {
}
}
constexpr auto FlagInfos = [] {
constexpr auto FlagInfos = std::invoke([] {
std::array<FlagInfo, OP_LAST> ret = {};
for (unsigned i = 0; i < OP_LAST; ++i) {
ret[i] = ClassifyConst(IROps(i));
ret[i] = ClassifyConst((IROps)i);
}
return ret;
}();
});
FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
FlagInfo Info = FlagInfos[IROp->Op];
@@ -352,22 +351,22 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
switch (IROp->Op) {
case OP_NZCVSELECT:
case OP_NZCVSELECTINCREMENT: {
auto Op = IROp->C<IR::IROp_NZCVSelect>();
auto Op = IROp->CW<IR::IROp_NZCVSelect>();
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
}
case OP_NZCVSELECTV: {
auto Op = IROp->C<IR::IROp_NZCVSelectV>();
auto Op = IROp->CW<IR::IROp_NZCVSelectV>();
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
}
case OP_NEG: {
auto Op = IROp->C<IR::IROp_Neg>();
auto Op = IROp->CW<IR::IROp_Neg>();
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
}
case OP_CONDJUMP: {
auto Op = IROp->C<IR::IROp_CondJump>();
auto Op = IROp->CW<IR::IROp_CondJump>();
if (!Op->FromNZCV) {
return FlagInfo::Pack({});
}
@@ -377,7 +376,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
case OP_CONDSUBNZCV:
case OP_CONDADDNZCV: {
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
auto Op = IROp->CW<IR::IROp_CondAddNZCV>();
return FlagInfo::Pack({
.Read = FlagsForCondClassType(Op->Cond),
.Write = FLAG_NZCV,
@@ -386,7 +385,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
}
case OP_RMIFNZCV: {
auto Op = IROp->C<IR::IROp_RmifNZCV>();
auto Op = IROp->CW<IR::IROp_RmifNZCV>();
static_assert(FLAG_N == (1 << 3), "rmif mask lines up with our bits");
static_assert(FLAG_Z == (1 << 2), "rmif mask lines up with our bits");
@@ -400,7 +399,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
}
case OP_INVALIDATEFLAGS: {
auto Op = IROp->C<IR::IROp_InvalidateFlags>();
auto Op = IROp->CW<IR::IROp_InvalidateFlags>();
unsigned Flags = 0;
// TODO: Make this translation less silly
@@ -537,7 +536,7 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie
// Initialize the FlagsRead mask according to the exit instruction.
auto [ExitNode, ExitOp] = CodeLast();
if (ExitOp->Op == IR::OP_CONDJUMP) {
auto Op = ExitOp->C<IR::IROp_CondJump>();
auto Op = ExitOp->CW<IR::IROp_CondJump>();
FlagsRead = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags;
} else if (ExitOp->Op == IR::OP_JUMP) {
FlagsRead = CFG.Get(ExitOp->Args[0])->Flags;
@@ -644,7 +643,7 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
for (auto [CodeNode, IROp] : CurrentIR.GetCode(Block)) {
if (IROp->Op == OP_STOREPF) {
auto Op = IROp->C<IR::IROp_StorePF>();
auto Op = IROp->CW<IR::IROp_StorePF>();
auto Generator = CurrentIR.GetOp<IR::IROp_Header>(Op->Value);
// Determine if we only write 0/1 to the parity flag.
@@ -697,7 +696,7 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
--CodeLast;
auto [ExitNode, ExitOp] = CodeLast();
if (ExitOp->Op == IR::OP_CONDJUMP) {
auto Op = ExitOp->C<IR::IROp_CondJump>();
auto Op = ExitOp->CW<IR::IROp_CondJump>();
CFG.RecordEdge(Block->ID, Op->TrueBlock);
CFG.RecordEdge(Block->ID, Op->FalseBlock);
@@ -748,7 +747,7 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
}
}
fextl::unique_ptr<Pass> CreateDeadFlagCalculationEliminination() {
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination() {
return fextl::make_unique<DeadFlagCalculationEliminination>();
}
Loaded 100 of 728 files, more files were not shown because too many files have changed in this diff. Show more