Compare commits

..
1 Commits
Author SHA1 Message Date
Ryan Houdek 33fe6813fc Docs: Update for release FEX-2104 2021-04-02 11:35:29 -07:00
994 changed files with 37590 additions and 112658 deletions

No files matched your search

@@ -1,45 +0,0 @@
---
name: Potential Game Bug
about: A bug in FEX-Emu that causes a problem in a game
title: "[Game]: [Short Problem Description]"
labels: Game related
assignees: ''
---
**What Game**
The game name.
A link to the storefront where to get the game. GOG, Steam, Itch.io, etc
**Describe the bug**
A clear and concise description of what the bug is.
**To Reproduce**
Steps to reproduce the behavior:
1. Go to '...'
2. Click on '....'
3. Scroll down to '....'
4. See error
**Expected behavior**
A clear and concise description of what you expected to happen.
**Screenshots and Video**
If applicable, add screenshots and video to help explain your problem.
**System information:**
- OS: [eg: Ubuntu 21.10]
- CPU/SoC: [eg: Snapdragon 888, Intel Core i8-12900k]
- Video driver version: [eg: OpenGL ES 3.2 Mesa 22.0.0-devel (git-9ff086052a)]
- RootFS used: [eg: Ubuntu 21.10 Official Rootfs]
- FEX version: (FEXGetConfig --version) [eg: FEX-2112-155-gc691d709]
- Thunks Enabled: [Yes/No]
**Additional context**
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
- Is this a Vulkan game: [Yes/No/Unknown]
- If Yes, What is your Vulkan driver:
Add any other context about the problem here.
+4 -98
View File
@@ -13,14 +13,13 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
jobs:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2], [self-hosted, ARMv8.4]]
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2]]
fail-fast: false
steps:
@@ -29,24 +28,9 @@ jobs:
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
run: git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
@@ -64,7 +48,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -132,18 +116,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
- name: gcc target tests 32
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
- name: GCC32 Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
- name: Struct verifier tests
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -155,73 +127,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
- name: APITest tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target api_tests
- name: APITest Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
- name: FEXLinuxTests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
- name: FEXLinuxTests Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
- name: Thunkgen tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target thunkgen_tests
- name: Thunkgen Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
- name: Install
if: matrix.arch[1] == 'x64'
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: Test GL No-Thunks
if: matrix.arch[1] == 'x64'
working-directory: ${{runner.workspace}}/build
shell: bash
env:
DISPLAY: ":0"
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_nothunks
- name: No thunks Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_NoThunkResults.log || true
- name: Test GL Thunks
if: matrix.arch[1] == 'x64'
working-directory: ${{runner.workspace}}/build
shell: bash
env:
DISPLAY: ":0"
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_thunks
- name: Thunks Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkResults.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
@@ -241,3 +146,4 @@ jobs:
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
-118
View File
@@ -1,118 +0,0 @@
name: Vixl Simulator run
on:
push:
branches:
- main
pull_request:
branches:
- main
env:
# Customize the CMake build type here (Release, Debug, RelWithDebInfo, etc.)
BUILD_TYPE: Release
CC: clang
CXX: clang++
jobs:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
# Only the x86-64 runner is fast enough to run this
arch: [[self-hosted, x64], [self-hosted, ARMv8.4]]
fail-fast: false
steps:
- uses: actions/checkout@v2
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: 'actions/upload-artifact@v2'
with:
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
-1
View File
@@ -10,4 +10,3 @@ out/
.vscode/
.vs/
*.pyc
.cache
+1 -24
View File
@@ -1,7 +1,7 @@
[submodule "External/vixl"]
shallow = true
path = External/vixl
url = https://github.com/FEX-Emu/vixl.git
url = https://github.com/Sonicadvance1/vixl.git
[submodule "External/cpp-optparse"]
path = External/cpp-optparse
url = https://github.com/Sonicadvance1/cpp-optparse
@@ -30,26 +30,3 @@
shallow = true
path = External/fex-gcc-target-tests-bins
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
[submodule "External/jemalloc"]
path = External/jemalloc
url = https://github.com/FEX-Emu/jemalloc.git
[submodule "External/fmt"]
path = External/fmt
url = https://github.com/fmtlib/fmt.git
[submodule "External/drm-headers"]
path = External/drm-headers
url = https://github.com/FEX-Emu/drm-headers.git
[submodule "External/xxhash"]
path = External/xxhash
url = https://github.com/FEX-Emu/xxHash.git
[submodule "External/Catch2"]
path = External/Catch2
url = https://github.com/catchorg/Catch2.git
[submodule "External/robin-map"]
shallow = true
path = External/robin-map
url = https://github.com/Tessil/robin-map.git
[submodule "External/Vulkan-Headers"]
shallow = true
path = External/Vulkan-Headers
url = https://github.com/KhronosGroup/Vulkan-Headers.git
-5
View File
@@ -1,5 +0,0 @@
{
"ThunksDB": {
"GL": 1
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"ThunksDB": {
"Vulkan": 1
}
}
-21
View File
@@ -1,21 +0,0 @@
if(NOT EXISTS "@CMAKE_BINARY_DIR@/install_manifest.txt")
message(FATAL_ERROR "Cannot find install manifest: @CMAKE_BINARY_DIR@/install_manifest.txt")
endif()
file(READ "@CMAKE_BINARY_DIR@/install_manifest.txt" files)
string(REGEX REPLACE "\n" ";" files "${files}")
foreach(file ${files})
message(STATUS "Uninstalling $ENV{DESTDIR}${file}")
if(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
exec_program(
"@CMAKE_COMMAND@" ARGS "-E remove \"$ENV{DESTDIR}${file}\""
OUTPUT_VARIABLE rm_out
RETURN_VALUE rm_retval
)
if(NOT "${rm_retval}" STREQUAL 0)
message(FATAL_ERROR "Problem when removing $ENV{DESTDIR}${file}")
endif()
else(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
message(STATUS "File $ENV{DESTDIR}${file} does not exist.")
endif()
endforeach()
+73 -340
View File
@@ -1,68 +1,24 @@
cmake_minimum_required(VERSION 3.14)
project(FEX)
INCLUDE (CheckIncludeFiles)
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
option(BUILD_THUNKS "Build thunks" FALSE)
option(BUILD_CLANG_THUNKS "Build thunks with clang" FALSE)
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
option(ENABLE_IWYU "Enables include what you use program" FALSE)
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
option(ENABLE_LLD "Enable linking with lld" FALSE)
option(ENABLE_MOLD "Enable linking with mold" FALSE)
option(ENABLE_LLD "Enable linking with LLD" FALSE)
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
option(ENABLE_WERROR "Enables -Werror" FALSE)
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
if (ENABLE_FEXCORE_PROFILER)
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
else()
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
endif()
endif()
# uninstall target
if(NOT TARGET uninstall)
configure_file(
"${CMAKE_CURRENT_SOURCE_DIR}/CMakeFiles/cmake_uninstall.cmake.in"
"${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake"
IMMEDIATE @ONLY)
add_custom_target(uninstall
COMMAND ${CMAKE_COMMAND} -P ${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake)
endif()
# These options are meant for package management
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
set (OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version in the format of <MMYY>{.<REV>}")
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
if (CMAKE_BUILD_TYPE MATCHES "DEBUG")
set(ENABLE_ASSERTIONS TRUE)
@@ -73,17 +29,6 @@ if (ENABLE_ASSERTIONS)
add_definitions(-DASSERTIONS_ENABLED=1)
endif()
if (ENABLE_GDB_SYMBOLS)
message(STATUS "GDBSymbols support enabled")
add_definitions(-DGDB_SYMBOLS_ENABLED=1)
endif()
if (ENABLE_INTERPRETER)
message(STATUS "Interpreter enabled")
add_definitions(-DINTERPRETER_ENABLED=1)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
@@ -99,6 +44,38 @@ else()
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION FALSE)
endif()
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
message(STATUS "CCache enabled")
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
endif()
if (ENABLE_XRAY)
add_compile_options(-fxray-instrument)
link_libraries(-fxray-instrument)
endif()
if (ENABLE_LLD)
link_libraries(-fuse-ld=lld)
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address)
link_libraries(-fno-omit-frame-pointer -fsanitize=address)
endif()
if (ENABLE_TSAN)
add_compile_options(-fno-omit-frame-pointer -fsanitize=thread)
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
endif()
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
@@ -113,89 +90,11 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
set(_M_ARM_64 1)
add_definitions(-D_M_ARM_64=1)
endif()
if (ENABLE_CCACHE)
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
message(STATUS "CCache enabled")
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
endif()
endif()
if (ENABLE_XRAY)
add_compile_options(-fxray-instrument)
link_libraries(-fxray-instrument)
endif()
if (ENABLE_COMPILE_TIME_TRACE)
add_compile_options(-ftime-trace)
link_libraries(-ftime-trace)
endif()
set (PTHREAD_LIB pthread)
if (ENABLE_LLD AND ENABLE_MOLD)
message (FATAL_ERROR "Cannot enable both lld and mold")
elseif (ENABLE_LLD)
set (LD_OVERRIDE "-fuse-ld=lld")
add_link_options(${LD_OVERRIDE})
elseif (ENABLE_MOLD)
add_link_options("-fuse-ld=mold")
endif()
if (ENABLE_LIBCXX)
message(WARNING "This is an unsupported configuration and should only be used for testing")
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -stdlib=libc++")
set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -stdlib=libc++ -lc++abi")
endif()
if (NOT ENABLE_OFFLINE_TELEMETRY)
# Disable FEX offline telemetry entirely if asked
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
endif()
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
add_definitions(-DTERMUX_BUILD=1)
set(TERMUX_BUILD 1)
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
set(ENABLE_JEMALLOC FALSE)
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
link_libraries(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
endif()
if (ENABLE_TSAN)
add_compile_options(-fno-omit-frame-pointer -fsanitize=thread)
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
endif()
if (ENABLE_JEMALLOC)
add_definitions(-DENABLE_JEMALLOC=1)
add_subdirectory(External/jemalloc/)
include_directories(External/jemalloc/pregen/include/)
else()
message (STATUS
" jemalloc disabled!\n"
" This is not a recommended configuration!\n"
" This will very explicitly break 32-bit application execution!\n"
" Use at your own risk!")
endif()
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
include_directories(External/robin-map/include/)
add_subdirectory(External/vixl/)
include_directories(External/vixl/src/)
@@ -204,30 +103,14 @@ if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
add_subdirectory(External/xxhash/)
include_directories(External/xxhash/)
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
if (BUILD_TESTS)
option(CATCH_BUILD_STATIC_LIBRARY "" ON)
set(CATCH_BUILD_STATIC_LIBRARY ON)
add_subdirectory(External/Catch2/)
# Pull in catch_discover_tests definition
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
include(Catch)
endif()
add_subdirectory(External/cpp-optparse/)
include_directories(External/cpp-optparse/)
add_subdirectory(External/fmt/)
add_subdirectory(External/imgui/)
include_directories(External/imgui/)
@@ -261,6 +144,11 @@ if(ENUM_ENUM_WARNING)
add_compile_options(-Wno-deprecated-enum-enum-conversion)
endif()
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
if(COMPILER_SUPPORTS_MARCH_NATIVE)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -march=native")
endif()
if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
add_compile_options(-Werror)
if (NOT ENABLE_STRICT_WERROR)
@@ -269,51 +157,19 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
endif()
endif()
if (NOT TUNE_ARCH STREQUAL "generic")
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
if(COMPILER_SUPPORTS_ARCH_TYPE)
add_compile_options("-march=${TUNE_ARCH}")
else()
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
endif()
endif()
if(_M_ARM_64)
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
# Manually detect newer CPU revisions until clang and llvm fixes their bug
# This script will either provide a supported CPU or 'native'
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo"
OUTPUT_VARIABLE AARCH64_CPU)
if (TUNE_CPU STREQUAL "native")
if(_M_ARM_64)
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
add_compile_options("-mcpu=native")
endif()
else()
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
# Manually detect newer CPU revisions until clang and llvm fixes their bug
# This script will either provide a supported CPU or 'native'
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
OUTPUT_VARIABLE AARCH64_CPU)
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
add_compile_options("-mcpu=${AARCH64_CPU}")
endif()
endif()
else()
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
if(COMPILER_SUPPORTS_MARCH_NATIVE)
add_compile_options("-march=native")
endif()
endif()
else()
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
add_compile_options("-mcpu=${TUNE_CPU}")
else()
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=${AARCH64_CPU}")
endif()
endif()
@@ -380,181 +236,58 @@ add_compile_options(-Wall)
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/include/Config.h.in
${CMAKE_BINARY_DIR}/generated/ConfigDefines.h)
${CMAKE_BINARY_DIR}/generated/Config.h)
if (BUILD_TESTS)
include(CTest)
enable_testing()
message(STATUS "Unit tests are enabled")
endif()
add_subdirectory(FEXHeaderUtils/)
add_subdirectory(External/FEXCore)
# Binfmt_misc files must be installed prior to Source/ installs
add_subdirectory(Data/binfmts/)
add_subdirectory(Source/)
add_subdirectory(Data/AppConfig/)
# Install the ThunksDB file
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/)
endforeach()
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
if (BUILD_THUNKS)
set (FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
add_subdirectory(ThunkLibs/Generator)
# Thunk targets for both host libraries and IDE integration
add_subdirectory(ThunkLibs/HostLibs)
# Thunk targets for IDE integration of guest code, only
add_subdirectory(ThunkLibs/GuestLibs)
# Thunk targets for guest libraries
include(ExternalProject)
ExternalProject_Add(host-libs
PREFIX host-libs
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/HostLibs"
BINARY_DIR "Host"
CMAKE_ARGS "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
)
install(
CODE "MESSAGE(\"-- Installing: host-libs\")"
CODE "
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkHostsInstall
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Host
)"
DEPENDS host-libs
)
ExternalProject_Add(guest-libs
PREFIX guest-libs
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
BINARY_DIR "Guest"
CMAKE_ARGS
"-DBITNESS=64"
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_64_TOOLCHAIN_FILE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
CMAKE_ARGS "-DX86_C_COMPILER:STRING=${X86_C_COMPILER}" "-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}" "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
DEPENDS thunkgen
)
ExternalProject_Add(guest-libs-32
PREFIX guest-libs-32
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
BINARY_DIR "Guest_32"
CMAKE_ARGS
"-DBITNESS=32"
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_32_TOOLCHAIN_FILE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
DEPENDS thunkgen
)
install(
CODE "MESSAGE(\"-- Installing: guest-libs\")"
CODE "
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkGuestsInstall
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)"
DEPENDS guest-libs
)
install(
CODE "MESSAGE(\"-- Installing: guest-libs-32\")"
CODE "
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)"
DEPENDS guest-libs-32
)
add_custom_target(uninstall_guest-libs
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)
add_custom_target(uninstall_guest-libs-32
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)
add_dependencies(uninstall uninstall_guest-libs)
add_dependencies(uninstall uninstall_guest-libs-32)
endif()
set(FEX_VERSION_MAJOR "0")
set(FEX_VERSION_MINOR "0")
set(FEX_VERSION_PATCH "0")
if (OVERRIDE_VERSION STREQUAL "detect")
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
RESULT_VARIABLE GIT_ERROR
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
if (NOT ${GIT_ERROR} EQUAL 0)
# Likely built in a way that doesn't have tags
# Setup a version tag that is unknown
set(GIT_DESCRIBE_STRING "FEX-0000")
endif()
endif()
else()
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
endif()
# Parse the version here
# Change something like `FEX-2106.1-76-<hash>` in to a list
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
# Extract the `2106.1` element
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
# Change `2106.1` in to a list
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
# Calculate list size
list(LENGTH DESCRIBE_LIST LIST_SIZE)
# Pull out the major version
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
# Minor version only exists if there is a .1 at the end
# eg: 2106 versus 2106.1
if (LIST_SIZE GREATER 1)
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
endif()
# Package creation
set (CPACK_GENERATOR "DEB")
set (CPACK_PACKAGE_NAME fex-emu)
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/CPack/Description.txt")
# Debian defines
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
"${CMAKE_CURRENT_SOURCE_DIR}/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/CPack/triggers")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
# binfmt_misc conflicts with qemu-user-static
# We also only install binfmt_misc on aarch64 hosts
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
endif()
include (CPack)
+1 -1
View File
@@ -55,7 +55,7 @@ further defined and clarified by project maintainers.
## Enforcement
Instances of abusive, harassing, or otherwise unacceptable behavior may be
reported by contacting the project team at team@fex-emu.com. All
reported by contacting the project team at team@fex-emu.org. All
complaints will be reviewed and investigated and will result in a response that
is deemed necessary and appropriate to the circumstances. The project team is
obligated to maintain confidentiality with regard to the reporter of an incident.
-3
View File
@@ -1,3 +0,0 @@
x86 and x86-64 Linux emulator
FEX is very much work in progress, so expect things to change.
-18
View File
@@ -1,18 +0,0 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Setup binfmt_misc
update-binfmts --import FEX-x86
update-binfmts --import FEX-x86_64
}
# Install FEXInterpreter hardlink
# Needs to be done before setting up binfmt_misc
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
-17
View File
@@ -1,17 +0,0 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Uninstall
update-binfmts --unimport FEX-x86
update-binfmts --unimport FEX-x86_64
}
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
# Remove FEXInterpreter hardlink
unlink /usr/bin/FEXInterpreter
-1
View File
@@ -1 +0,0 @@
activate-noawait ldconfig
+3 -3
View File
@@ -11,15 +11,15 @@ endforeach()
# First generate then install it
foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
# Get the filename only component
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WLE)
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WE)
# Configure it
configure_file(
${GEN_CONFIG_SRC}
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json)
# Then install the configured json
install(
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"x86dec_SynchronizeRIPOnAllBlocks": "1"
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"x86dec_SynchronizeRIPOnAllBlocks": "1"
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"AdditionalArguments": "--no-sandbox"
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"x86dec_SynchronizeRIPOnAllBlocks": "1"
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"x86dec_SynchronizeRIPOnAllBlocks": "1"
}
}
-178
View File
@@ -1,178 +0,0 @@
{
"DB": {
"GL": {
"Library" : "libGL-guest.so",
"Depends": [
"X11"
],
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1",
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.2.0",
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.7.0"
]
},
"GLESv2": {
"Library": "libGLESv2-guest.so",
"Depends": [
"X11"
],
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2",
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2.0.0"
]
},
"X11": {
"Library": "libX11-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6",
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6.4.0"
]
},
"Vulkan": {
"Library": "libvulkan-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so.1",
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
],
"Comment": [
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
]
},
"xcb": {
"Library": "libxcb-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1.1.0"
]
},
"xcb-dri2": {
"Library": "libxcb_dri2-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
]
},
"xcb-dri3": {
"Library": "libxcb_dri3-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
]
},
"xcb-xfixes": {
"Library": "libxcb_xfixes-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
]
},
"xcb-shm": {
"Library": "libxcb_shm-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
]
},
"xcb-sync": {
"Library": "libxcb_sync-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
]
},
"xcb-randr": {
"Library": "libxcb_randr-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
]
},
"xcb-present": {
"Library": "libxcb_present-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0.0.0"
]
},
"xcb-glx": {
"Library": "libxcb_glx-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0",
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
]
},
"xshmfence": {
"Library": "libshmfence-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1",
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1.0.0"
]
},
"drm": {
"Library": "libdrm-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2",
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2.4.0"
]
},
"asound": {
"Library": "libasound-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2",
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2.0.0"
]
},
"Xrender": {
"Library": "libXrender-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1",
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1.3.0"
]
},
"Xext": {
"Library": "libXext-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6",
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6.4.0"
]
},
"Xfixes": {
"Library": "libXfixes-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3",
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3.1.0"
]
},
"OpenCL": {
"Library" : "libOpenCL-guest.so",
"Overlay": [
"@PREFIX_LIB@/x86_64-linux-gnu/libOpenCL.so",
"@PREFIX_LIB@/x86_64-linux-gnu/libOpenCL.so.1",
"@PREFIX_LIB@/x86_64-linux-gnu/libOpenCL.so.1.0.0"
]
},
"":{}
}
}
-17
View File
@@ -1,17 +0,0 @@
function(GenBinFmt Name)
# Get the filename only component
get_filename_component(FMT_NAME ${Name} NAME_WE)
# Configure it
configure_file(
${Name}
${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
# Then install the configured binfmt
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
endfunction()
GenBinFmt(FEX-x86.in)
GenBinFmt(FEX-x86_64.in)
-8
View File
@@ -1,8 +0,0 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
credentials yes
fix_binary yes
preserve yes
-8
View File
@@ -1,8 +0,0 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
credentials yes
fix_binary yes
preserve yes
+2 -2
View File
@@ -3,7 +3,7 @@ FROM ubuntu:20.04 as builder
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
clang-10 llvm-10 nasm ninja-build \
clang-10 llvm-10 nasm ninja-build libnuma-dev \
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
python3 linux-headers-generic
@@ -23,7 +23,7 @@ FROM ubuntu:20.04
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y \
libcap-dev libglfw3-dev libepoxy-dev
libnuma-dev libcap-dev libglfw3-dev libepoxy-dev
COPY --from=builder /opt/FEX/build/Bin/* /usr/bin/
-1
Submodule External/Catch2 deleted from c4e3767e26.
+23 -33
View File
@@ -9,20 +9,14 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
set(_M_ARM_64 1)
endif()
if (ENABLE_VIXL_SIMULATOR)
# If the vixl simulator is enabled then we are using the ARM64 JIT
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" FALSE)
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" TRUE)
else()
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" ${_M_X86_64})
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" ${_M_ARM_64})
endif()
set(ENABLE_JIT_X86_64 ${_M_X86_64} CACHE BOOL "Enable the x86_64 JIT")
set(ENABLE_JIT_ARM64 ${_M_ARM_64} CACHE BOOL "Enable the ARM64 JIT")
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
option(ENABLE_JITSYMBOLS "Enable visibility of JITSymbols in profiling tools" FALSE)
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
@@ -34,6 +28,7 @@ set(CMAKE_INCLUDE_CURRENT_DIR ON)
include(CheckCXXCompilerFlag)
include(CheckIncludeFileCXX)
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
# Useful to have for freestanding libFEXCore
add_subdirectory(External/vixl/)
@@ -43,32 +38,27 @@ endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
# Find our git hash
find_package(Git)
set(GIT_SHORT_HASH "Unknown")
set(GIT_DESCRIBE_STRING "FEX-Unknown")
if (OVERRIDE_VERSION STREQUAL "detect")
# Find our git hash
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} rev-parse --short=7 HEAD
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_SHORT_HASH
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
endif()
else()
set(GIT_SHORT_HASH "${OVERRIDE_VERSION}")
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} rev-parse --short HEAD
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_SHORT_HASH
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
endif()
configure_file(
+12 -2
View File
@@ -18,8 +18,14 @@ This project aims to provide a fast and functional x86-64 emulation library that
* Portable library implementation in order to support easy integration in to applications
### Target Host Architecture
The target host architecture for this library is AArch64. Specifically the ARMv8.1 version or newer.
The CPU IR is designed with AArch64 in mind but should allow for other architectures as well.
x86-64 host support is available for ease of development, but is not a priority.
The CPU IR is designed with AArch64 in mind but there is a desire to run the recompiled code on other architectures as well.
Multiple architecture support is desired for easier bringup and debugging, performance isn't as much of a priority there (ex. x86-64(guest) translated to x86-64(host))
### Not currently goals but will be in the future
* 32bit x86 support
* This will be a desire in the future, but to lower the amount of work required, decided to push this off for now.
* Integration in to WINE
* Later generation of x86-64 instruction sets
* Including AVX, F16C, XOP, FMA, AVX2, etc
### Not desired
* Kernel space emulation
* CPL0-2 emulation
@@ -27,3 +33,7 @@ x86-64 host support is available for ease of development, but is not a priority.
* IRQs
* SVM
* "Cycle Accurate" emulation
### Dependencies
* clang-tidy if you want to ensure the code stays tidy
* cmake
* A C++17 compliant compiler (There are assumptions made about using Clang and LTO)
+5 -65
View File
@@ -98,19 +98,14 @@ def print_man_option(short, long, desc, default):
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
def print_man_env_option(name, desc, default, no_json_key):
output_man.write("\\fBFEX_{0}\\fR\n".format(name.upper()))
def print_man_env_option(name, desc, default):
output_man.write("\\fBFEX_{0}\\fR\n".format(name))
# Print description
for line in desc:
output_man.write(".Pp\n")
output_man.write("{0}\n".format(line))
if (not no_json_key):
output_man.write(".Pp\n")
output_man.write("\\fBJSON key:\\fR '{0}'\n".format(name))
output_man.write(".Pp\n\n")
output_man.write(".Pp\n")
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
@@ -159,48 +154,12 @@ def print_man_environment(options):
# Wrap the string argument in quotes
default = "'" + default + "'"
print_man_env_option(
op_key,
op_key.upper(),
op_vals["Desc"],
default,
False
default
)
print_man_environment_tail()
output_man.write(".El\n")
def print_man_environment_tail():
# Additional environment variables that live outside of the normal loop
print_man_env_option(
"FEX_APP_CONFIG_LOCATION",
[
"Allows the user to override where FEX looks for configuration files",
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
"This will override the full path",
],
"''", True)
print_man_env_option(
"FEX_APP_CONFIG",
[
"Allows the user to override where FEX looks for only the application config file",
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/Config.json",
"This will override this file location",
"One must be careful with this option as it will override any applications that load with execve as well"
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
],
"''", True)
print_man_env_option(
"FEX_APP_DATA_LOCATION",
[
"Allows the user to override where FEX looks for data files",
"By default FEX will look in {$HOME, $XDG_DATA_HOME}/.fex-emu/",
"This will override the full path",
"This is the folder where FEX stores generated files like IR cache"
],
"''", True)
def print_man_header():
header ='''.Dd {0}
.Dt FEX
@@ -374,7 +333,7 @@ def print_parse_argloader_options(options):
conversion_func = "std::to_string"
if ("ArgumentHandler" in op_vals):
NeedsString = True
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
conversion_func = "FEX::Handler::{0}".format(op_vals["ArgumentHandler"])
if (value_type == "str"):
NeedsString = True
conversion_func = ""
@@ -396,21 +355,6 @@ def print_parse_argloader_options(options):
output_argloader.write("#endif\n")
def print_parse_envloader_options(options):
output_argloader.write("#ifdef ENVLOADER\n")
output_argloader.write("#undef ENVLOADER\n")
output_argloader.write("if (false) {}\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
if ("ArgumentHandler" in op_vals):
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
output_argloader.write("Value = {0}(Value);\n".format(conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def check_for_duplicate_options(options):
short_map = []
long_map = []
@@ -485,8 +429,4 @@ output_man.close()
output_argloader = open(output_argumentloader_filename, "w")
print_argloader_options(options);
print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
output_argloader.close()
+43 -6
View File
@@ -7,12 +7,17 @@ OpClasses = collections.OrderedDict()
def get_ir_classes(ops, defines):
global OpClasses
for op_class, opslist in ops.items():
if not (op_class in OpClasses):
OpClasses[op_class] = []
for op_key, op_vals in ops.items():
if not ("Last" in op_vals):
OpClass = "#Unknown"
for op, op_val in opslist.items():
OpClasses[op_class].append([op, op_val])
if ("OpClass" in op_vals):
OpClass = op_vals["OpClass"]
if not (OpClass in OpClasses):
OpClasses[OpClass] = []
OpClasses[OpClass].append([op_key, op_vals])
# Sort the dictionary after we are done parsing it
OpClasses = collections.OrderedDict(sorted(OpClasses.items()))
@@ -33,9 +38,41 @@ def print_ir_ops():
op_key = op[0]
op_vals = op[1]
output_file.write("## %s\n" % (op_key))
HasDest = ("HasDest" in op_vals and op_vals["HasDest"] == True)
HasSSAArgs = ("SSAArgs" in op_vals and len(op_vals["SSAArgs"]) > 0)
HasSSAArgNames = "SSANames" in op_vals
HasArgs = "Args" in op_vals
SSAArgsCount = 0
ArgCount = 0
if (HasSSAArgs):
SSAArgsCount = int(op_vals["SSAArgs"])
if (HasArgs):
ArgCount = len(op_vals["Args"])
TotalArgsCount = SSAArgsCount + (ArgCount / 2)
output_file.write(">")
output_file.write(op_key)
if (HasDest):
output_file.write("%dest = ")
output_file.write("%s " % op_key)
ArgComma = (", ", "")
if (HasSSAArgs):
for i in range(0, SSAArgsCount):
FinalArg = (i + 1) == TotalArgsCount
if (HasSSAArgNames):
output_file.write("%%%s%s" % (op_vals["SSANames"][i], ArgComma[FinalArg]))
else:
output_file.write("%%ssa%d%s" % (i, ArgComma[FinalArg]))
if (HasArgs):
Args = op_vals["Args"]
for i in range(0, ArgCount, 2):
FinalArg = ((i / 2) + SSAArgsCount + 1) == TotalArgsCount
data_type = Args[i]
data_name = Args[i + 1]
output_file.write("\<%s %s\>%s" % (data_type, data_name, ArgComma[FinalArg]))
output_file.write("\n\n")
Vendored Executable → Regular
+395 -492
View File
File diff suppressed because it is too large. Load diff
+37 -116
View File
@@ -1,15 +1,9 @@
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
set (FEXCORE_BASE_SRCS
Common/Paths.cpp
Interface/Config/Config.cpp
Utils/FileLoading.cpp
Utils/ForcedAssert.cpp
Utils/LogManager.cpp
)
set (SRCS
Common/Paths.cpp
Common/JitSymbols.cpp
Common/NetStream.cpp
Common/SoftFloat-3e/extF80_add.c
Common/SoftFloat-3e/extF80_div.c
Common/SoftFloat-3e/extF80_sub.c
@@ -77,25 +71,17 @@ set (SRCS
Common/SoftFloat-3e/f32_to_extF80.c
Common/SoftFloat-3e/s_normSubnormalF32Sig.c
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
Interface/Config/Config.cpp
Interface/Context/Context.cpp
Interface/Core/LookupCache.cpp
Interface/Core/BlockSamplingData.cpp
Interface/Core/CompileService.cpp
Interface/Core/Core.cpp
Interface/Core/CPUBackend.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/GdbServer.cpp
Interface/Core/HostFeatures.cpp
Interface/Core/ObjectCache/JobHandling.cpp
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
Interface/Core/ObjectCache/ObjectCacheService.cpp
Interface/Core/OpcodeDispatcher/Crypto.cpp
Interface/Core/OpcodeDispatcher/Flags.cpp
Interface/Core/OpcodeDispatcher/Vector.cpp
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher/X87F64.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/SignalDelegator.cpp
Interface/Core/X86Tables.cpp
Interface/Core/X86DebugInfo.cpp
Interface/Core/X86HelperGen.cpp
@@ -104,7 +90,8 @@ set (SRCS
Interface/Core/Dispatcher/Dispatcher.cpp
Interface/Core/Dispatcher/X86Dispatcher.cpp
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
Interface/Core/Interpreter/InterpreterFallbacks.cpp
Interface/Core/Interpreter/InterpreterCore.cpp
Interface/Core/Interpreter/InterpreterOps.cpp
Interface/Core/X86Tables/BaseTables.cpp
Interface/Core/X86Tables/DDDTables.cpp
Interface/Core/X86Tables/EVEXTables.cpp
@@ -118,8 +105,6 @@ set (SRCS
Interface/Core/X86Tables/X87Tables.cpp
Interface/Core/X86Tables/XOPTables.cpp
Interface/HLE/Thunks/Thunks.cpp
Interface/GDBJIT/GDBJIT.cpp
Interface/IR/AOTIR.cpp
Interface/IR/IRDumper.cpp
Interface/IR/IRParser.cpp
Interface/IR/IREmitter.cpp
@@ -129,8 +114,6 @@ set (SRCS
Interface/IR/Passes/DeadContextStoreElimination.cpp
Interface/IR/Passes/IRCompaction.cpp
Interface/IR/Passes/IRValidation.cpp
Interface/IR/Passes/RAValidation.cpp
Interface/IR/Passes/LongDivideRemovalPass.cpp
Interface/IR/Passes/ValueDominanceValidation.cpp
Interface/IR/Passes/PhiValidation.cpp
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
@@ -138,37 +121,18 @@ set (SRCS
Interface/IR/Passes/StaticRegisterAllocationPass.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/SyscallOptimization.cpp
Utils/Allocator.cpp
Utils/Allocator/64BitAllocator.cpp
Utils/NetStream.cpp
Utils/Telemetry.cpp
Utils/ELFLoader.cpp
Utils/ELFSymbolDatabase.cpp
Utils/LogManager.cpp
Utils/Threads.cpp
Utils/Profiler.cpp
)
if (ENABLE_INTERPRETER)
list(APPEND SRCS
Interface/Core/Interpreter/InterpreterCore.cpp
Interface/Core/Interpreter/InterpreterOps.cpp
Interface/Core/Interpreter/ALUOps.cpp
Interface/Core/Interpreter/AtomicOps.cpp
Interface/Core/Interpreter/BranchOps.cpp
Interface/Core/Interpreter/ConversionOps.cpp
Interface/Core/Interpreter/EncryptionOps.cpp
Interface/Core/Interpreter/F80Ops.cpp
Interface/Core/Interpreter/FlagOps.cpp
Interface/Core/Interpreter/MemoryOps.cpp
Interface/Core/Interpreter/MiscOps.cpp
Interface/Core/Interpreter/MoveOps.cpp
Interface/Core/Interpreter/VectorOps.cpp)
endif()
if(_M_ARM_64)
list(APPEND SRCS
Interface/Core/ArchHelpers/Arm64.cpp)
endif()
set(DEFINES -DTHREAD_LOCAL=_Thread_local)
set(DEFINES )
if (_M_X86_64)
list(APPEND DEFINES -D_M_X86_64=1)
@@ -178,11 +142,6 @@ if (_M_ARM_64)
list(APPEND DEFINES -D_M_ARM_64=1)
endif()
if (ENABLE_VIXL_SIMULATOR)
# We can run the simulator on both x86-64 or AArch64 hosts
list(APPEND DEFINES -DVIXL_SIMULATOR=1 -DVIXL_INCLUDE_SIMULATOR_AARCH64=1)
endif()
if (ENABLE_JIT_X86_64)
list(APPEND SRCS
Interface/Core/JIT/x86_64/JIT.cpp
@@ -195,9 +154,7 @@ if (ENABLE_JIT_X86_64)
Interface/Core/JIT/x86_64/MemoryOps.cpp
Interface/Core/JIT/x86_64/MiscOps.cpp
Interface/Core/JIT/x86_64/MoveOps.cpp
Interface/Core/JIT/x86_64/VectorOps.cpp
Interface/Core/JIT/x86_64/x64Relocations.cpp
)
Interface/Core/JIT/x86_64/VectorOps.cpp)
list(APPEND DEFINES -DJIT_X86_64)
endif()
@@ -214,31 +171,25 @@ if (ENABLE_JIT_ARM64)
Interface/Core/JIT/Arm64/MemoryOps.cpp
Interface/Core/JIT/Arm64/MiscOps.cpp
Interface/Core/JIT/Arm64/MoveOps.cpp
Interface/Core/JIT/Arm64/VectorOps.cpp
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
)
Interface/Core/JIT/Arm64/VectorOps.cpp)
endif()
set (LIBS fmt::fmt vixl dl xxhash tiny-json FEXHeaderUtils)
if (ENABLE_JEMALLOC)
list (APPEND LIBS FEX_jemalloc)
if (ENABLE_JITSYMBOLS)
list(APPEND DEFINES -DENABLE_JITSYMBOLS=1)
endif()
# Generate config
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
${CMAKE_BINARY_DIR}/generated/Config/Config.json)
# Generate IR include file
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
add_custom_target(CREATE_IR_FOLDER ALL
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_IR_FOLDER}")
add_custom_command(
OUTPUT "${OUTPUT_NAME}"
DEPENDS "${INPUT_NAME}"
DEPENDS CREATE_IR_FOLDER
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
)
@@ -252,6 +203,7 @@ set(OUTPUT_IR_DOC "${CMAKE_BINARY_DIR}/IR.md")
add_custom_command(
OUTPUT "${OUTPUT_IR_DOC}"
DEPENDS "${INPUT_NAME}"
DEPENDS CREATE_IR_FOLDER
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py" "${INPUT_NAME}" "${OUTPUT_IR_DOC}"
)
@@ -268,28 +220,23 @@ add_custom_target(IR_INC
set(OUTPUT_CONFIG_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/Config")
set(OUTPUT_CONFIG_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigValues.inl")
set(OUTPUT_CONFIG_OPTION_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigOptions.inl")
set(INPUT_CONFIG_NAME "${CMAKE_BINARY_DIR}/generated/Config/Config.json")
set(INPUT_CONFIG_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json")
set(OUTPUT_MAN_NAME "${CMAKE_BINARY_DIR}/generated/FEX.1")
set(OUTPUT_MAN_NAME_COMPRESS "${CMAKE_BINARY_DIR}/generated/FEX.1.gz")
file(MAKE_DIRECTORY "${OUTPUT_CONFIG_FOLDER}")
add_custom_target(CREATE_CONFIG_FOLDER ALL
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_CONFIG_FOLDER}")
add_custom_command(
OUTPUT "${OUTPUT_CONFIG_NAME}"
OUTPUT "${OUTPUT_CONFIG_OPTION_NAME}"
OUTPUT "${OUTPUT_MAN_NAME}"
DEPENDS "${INPUT_CONFIG_NAME}"
DEPENDS CREATE_CONFIG_FOLDER
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py" "${INPUT_CONFIG_NAME}" "${OUTPUT_CONFIG_NAME}" "${OUTPUT_MAN_NAME}"
"${OUTPUT_CONFIG_OPTION_NAME}"
)
add_custom_command(
OUTPUT "${OUTPUT_MAN_NAME_COMPRESS}"
DEPENDS "${OUTPUT_MAN_NAME}"
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}"
)
set_source_files_properties(${OUTPUT_CONFIG_NAME} PROPERTIES
GENERATED TRUE)
set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
@@ -297,28 +244,29 @@ set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
set_source_files_properties(${OUTPUT_MAN_NAME} PROPERTIES
GENERATED TRUE)
set_source_files_properties(${OUTPUT_MAN_NAME_COMPRESS} PROPERTIES
GENERATED TRUE)
# Create the target
add_custom_target(CONFIG_INC
DEPENDS "${OUTPUT_CONFIG_NAME}"
DEPENDS "${OUTPUT_CONFIG_OPTION_NAME}"
DEPENDS "${OUTPUT_MAN_NAME}"
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
DEPENDS "${OUTPUT_MAN_NAME}")
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
# Install the man page
install(FILES ${OUTPUT_MAN_NAME} DESTINATION ${MAN_DIR}/man1)
# Add in diagnostic colours if the option is available.
# Ninja code generator will kill colours if this isn't here
check_cxx_compiler_flag(-fdiagnostics-color=always GCC_COLOR)
check_cxx_compiler_flag(-fcolor-diagnostics CLANG_COLOR)
function(AddDefaultOptionsToTarget Name)
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
set_target_properties(${Name} PROPERTIES VISIBILITY_INLINES_HIDDEN TRUE)
function(AddObject Name Type)
add_library(${Name} ${Type} ${SRCS})
add_dependencies(${Name} IR_INC)
add_dependencies(${Name} CONFIG_INC)
target_link_libraries(${Name} pthread rt vixl ${LINUX_LIBS} dl)
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
target_include_directories(${Name} PRIVATE IncludePrivate/)
@@ -328,18 +276,13 @@ function(AddDefaultOptionsToTarget Name)
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
target_compile_definitions(${Name} PRIVATE ${DEFINES})
add_dependencies(${Name} CONFIG_INC)
target_compile_options(${Name}
PRIVATE
-Wall
-Werror=cast-qual
-Werror=ignored-qualifiers
-Werror=implicit-fallthrough
-Wno-trigraphs
-ffunction-sections
-fwrapv
)
if (GCC_COLOR)
@@ -352,38 +295,16 @@ function(AddDefaultOptionsToTarget Name)
PRIVATE
"-fcolor-diagnostics")
endif()
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
target_link_options(${Name}
PRIVATE
"LINKER:--gc-sections"
"LINKER:--strip-all"
"LINKER:--as-needed"
)
endif()
endfunction()
# Build FEXCore_Config static library
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
target_link_libraries(FEXCore_Base ${LIBS})
AddDefaultOptionsToTarget(FEXCore_Base)
function(AddObject Name Type)
add_library(${Name} ${Type} ${SRCS})
add_dependencies(${Name} IR_INC)
target_link_libraries(${Name} FEXCore_Base)
AddDefaultOptionsToTarget(${Name})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
endfunction()
function(AddLibrary Name Type)
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
target_link_libraries(${Name} FEXCore_Base)
target_link_libraries(${Name} pthread rt vixl ${LINUX_LIBS} dl)
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
AddDefaultOptionsToTarget(${Name})
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
target_include_directories(${Name} PUBLIC "${PROJECT_SOURCE_DIR}/include/")
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
endfunction()
AddObject(${PROJECT_NAME}_object OBJECT)
+13 -18
View File
@@ -1,16 +1,12 @@
#pragma once
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/MathUtils.h>
#include "Common/MathUtils.h"
#include <FEXCore/Utils/LogManager.h>
#include <cstdint>
#include <cstdlib>
#include <cstring>
#include <stdint.h>
#include <stdlib.h>
#include <type_traits>
namespace FEXCore {
template<typename T>
struct BitSet final {
using ElementType = T;
@@ -20,16 +16,16 @@ struct BitSet final {
ElementType *Memory;
void Allocate(size_t Elements) {
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
LogMan::Throw::A((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(malloc(AllocateSize));
}
void Realloc(size_t Elements) {
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
LogMan::Throw::A((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(realloc(Memory, AllocateSize));
}
void Free() {
FEXCore::Allocator::free(Memory);
free(Memory);
Memory = nullptr;
}
bool Get(T Element) {
@@ -64,8 +60,8 @@ struct BitSetView final {
ElementType *Memory;
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0,
"Bitset view offset needs to be aligned to size of backing element");
LogMan::Throw::A((ElementOffset % MinimumSize) == 0,
"Bitset view offset needs to be aligned to size of backing element");
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
}
@@ -90,12 +86,11 @@ struct BitSetView final {
bool operator[](T Element) {
return Get(Element);
}
};
static_assert(sizeof(BitSet<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
static_assert(std::is_trivially_copyable_v<BitSet<uint32_t>>, "Needs to trivially copyable");
static_assert(std::is_trivially_copyable<BitSet<uint32_t>>::value, "Needs to trivially copyable");
static_assert(sizeof(BitSetView<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
static_assert(std::is_trivially_copyable_v<BitSetView<uint32_t>>, "Needs to trivially copyable");
} // namespace FEXCore
static_assert(std::is_trivially_copyable<BitSetView<uint32_t>>::value, "Needs to trivially copyable");
+21 -40
View File
@@ -1,64 +1,45 @@
#include "Common/JitSymbols.h"
#include <string>
#include <sstream>
#include <unistd.h>
#include <fmt/format.h>
namespace FEXCore {
JITSymbols::JITSymbols() : fp{nullptr, std::fclose} {
}
JITSymbols::JITSymbols() {
std::stringstream PerfMap;
PerfMap << "/tmp/perf-" << getpid() << ".map";
JITSymbols::~JITSymbols() = default;
void JITSymbols::InitFile() {
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
fp.reset(fopen(PerfMap.c_str(), "wb"));
fp = fopen(PerfMap.str().c_str(), "wb");
if (fp) {
// Disable buffering on this file
setvbuf(fp.get(), nullptr, _IONBF, 0);
setvbuf(fp, nullptr, _IONBF, 0);
}
}
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
JITSymbols::~JITSymbols() {
if (fp) {
fclose(fp);
}
}
void JITSymbols::Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
fmt::print(fp.get(), "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
std::stringstream String;
String << std::hex << HostAddr << " " << CodeSize << " " << "JIT_0x" << GuestAddr << "_" << HostAddr << std::endl;
fwrite(String.str().c_str(), 1, String.str().size(), fp);
}
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
void JITSymbols::Register(void *HostAddr, uint32_t CodeSize, std::string const &Name) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
fmt::print(fp.get(), "{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
std::stringstream String;
String << std::hex << HostAddr << " " << CodeSize << " " << Name << "_" << HostAddr << std::endl;
fwrite(String.str().c_str(), 1, String.str().size(), fp);
}
}
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
fmt::print(fp.get(), "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
}
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
fmt::print(fp.get(), "{} {:x} {}\n", HostAddr, CodeSize, Name);
}
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
fmt::print(fp.get(), "{} {:x} FEXJIT\n", HostAddr, CodeSize);
}
} // namespace FEXCore
+4 -13
View File
@@ -1,26 +1,17 @@
#pragma once
#include <cstdint>
#include <cstdio>
#include <memory>
#include <string_view>
#include <string>
namespace FEXCore {
class JITSymbols final {
public:
JITSymbols();
~JITSymbols();
void InitFile();
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
void Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(void *HostAddr, uint32_t CodeSize, std::string const &Name);
private:
using FILEPtr = std::unique_ptr<FILE, decltype(&std::fclose)>;
FILEPtr fp;
FILE* fp{};
};
}
+13
View File
@@ -0,0 +1,13 @@
#pragma once
#include <stdint.h>
static inline uint64_t AlignUp(uint64_t value, uint64_t size) {
return value + (size - value % size) % size;
};
static inline uint64_t AlignDown(uint64_t value, uint64_t size) {
return value - value % size;
};
@@ -1,47 +1,19 @@
#include <FEXCore/Utils/NetStream.h>
#include "NetStream.h"
#include <array>
#include <cstring>
#include <iterator>
#include <sys/types.h>
#include <sys/socket.h>
#include <stdio.h>
#include <unistd.h>
namespace FEXCore::Utils {
namespace {
class NetBuf final : public std::streambuf {
public:
explicit NetBuf(int socketfd) : socket{socketfd} {
reset_output_buffer();
}
~NetBuf() override {
close(socket);
}
private:
std::streamsize xsputn(const char* buffer, std::streamsize size) override;
std::streambuf::int_type underflow() override;
std::streambuf::int_type overflow(std::streambuf::int_type ch) override;
int sync() override;
void reset_output_buffer() {
// we always leave room for one extra char
setp(std::begin(output_buffer), std::end(output_buffer) -1);
}
int flushBuffer(const char *buffer, size_t size);
int socket;
std::array<char, 1400> output_buffer;
std::array<char, 1500> input_buffer; // enough for a typical packet
};
int NetBuf::flushBuffer(const char *buffer, size_t size) {
int NetStream::NetBuf::flushBuffer(const char *buffer, size_t size) {
size_t total = 0;
// Send data
while (total < size) {
size_t sent = send(socket, (const void*)(buffer + total), size - total, MSG_NOSIGNAL);
size_t sent = send(socket, (const void*)(buffer + total), size - total, 0);
if (sent == -1) {
// lets just assume all errors are end of file.
return -1;
@@ -52,12 +24,12 @@ int NetBuf::flushBuffer(const char *buffer, size_t size) {
return 0;
}
std::streamsize NetBuf::xsputn(const char* buffer, std::streamsize size) {
std::streamsize NetStream::NetBuf::xsputn(const char* buffer, std::streamsize size) {
size_t buf_remaining = epptr() - pptr();
// Check if the string fits neatly in our buffer
if (size <= buf_remaining) {
::memcpy(pptr(), buffer, size);
std::memcpy(pptr(), buffer, size);
pbump(size);
return size;
}
@@ -76,23 +48,23 @@ std::streamsize NetBuf::xsputn(const char* buffer, std::streamsize size) {
}
}
std::streambuf::int_type NetBuf::overflow(std::streambuf::int_type ch) {
std::streambuf::int_type NetStream::NetBuf::overflow(std::streambuf::int_type ch) {
// we always leave room for one extra char
*pptr() = (char) ch;
pbump(1);
return sync();
}
int NetBuf::sync() {
int NetStream::NetBuf::sync() {
// Flush and reset output buffer to zero
if (flushBuffer(pbase(), pptr() - pbase()) < 0) {
if(flushBuffer(pbase(), pptr() - pbase()) < 0) {
return -1;
}
reset_output_buffer();
return 0;
}
std::streambuf::int_type NetBuf::underflow() {
std::streambuf::int_type NetStream::NetBuf::underflow() {
ssize_t size = recv(socket, (void *)std::begin(input_buffer), sizeof(input_buffer), 0);
if (size <= 0) {
@@ -104,12 +76,11 @@ std::streambuf::int_type NetBuf::underflow() {
return traits_type::to_int_type(*gptr());
}
} // Anonymous namespace
NetStream::NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
NetStream::~NetStream() {
delete rdbuf();
}
} // namespace FEXCore::Utils
NetStream::NetBuf::~NetBuf() {
close(socket);
}
+41
View File
@@ -0,0 +1,41 @@
#pragma once
#include <array>
#include <iostream>
#include <string.h>
class NetStream : public std::iostream {
public:
NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
virtual ~NetStream();
private:
class NetBuf : public std::streambuf {
public:
NetBuf(int socketfd) {
socket = socketfd;
reset_output_buffer();
}
virtual ~NetBuf();
protected:
virtual std::streamsize xsputn(const char* buffer, std::streamsize size);
virtual std::streambuf::int_type underflow();
virtual std::streambuf::int_type overflow(std::streambuf::int_type ch);
virtual int sync();
private:
void reset_output_buffer() {
// we always leave room for one extra char
setp(std::begin(output_buffer), std::end(output_buffer) -1);
}
int flushBuffer(const char *buffer, size_t size);
int socket;
std::array<char, 1400> output_buffer;
std::array<char, 1500> input_buffer; // enough for a typical packet
};
};
+12 -53
View File
@@ -3,48 +3,13 @@
#include <cstdlib>
#include <filesystem>
#include <memory>
#include <pwd.h>
#include <system_error>
#include <unistd.h>
#include <sys/stat.h>
namespace FEXCore::Paths {
std::unique_ptr<std::string> CachePath;
std::unique_ptr<std::string> EntryCache;
char const* FindUserHomeThroughUID() {
auto passwd = getpwuid(geteuid());
if (passwd) {
return passwd->pw_dir;
}
return nullptr;
}
const char *GetHomeDirectory() {
char const *HomeDir = getenv("HOME");
// Try to get home directory from uid
if (!HomeDir) {
HomeDir = FindUserHomeThroughUID();
}
// try the PWD
if (!HomeDir) {
HomeDir = getenv("PWD");
}
// Still doesn't exit? You get local
if (!HomeDir) {
HomeDir = ".";
}
return HomeDir;
}
std::string CachePath;
std::string EntryCache;
void InitializePaths() {
CachePath = std::make_unique<std::string>();
EntryCache = std::make_unique<std::string>();
char const *HomeDir = getenv("HOME");
if (!HomeDir) {
@@ -57,35 +22,29 @@ namespace FEXCore::Paths {
char *XDGDataDir = getenv("XDG_DATA_DIR");
if (XDGDataDir) {
*CachePath = XDGDataDir;
CachePath = XDGDataDir;
}
else {
if (HomeDir) {
*CachePath = HomeDir;
CachePath = HomeDir;
}
}
*CachePath += "/.fex-emu/";
*EntryCache = *CachePath + "/EntryCache/";
CachePath += "/.fex-emu/";
EntryCache = CachePath + "/EntryCache/";
std::error_code ec{};
// Ensure the folder structure is created for our Data
if (!std::filesystem::exists(*EntryCache, ec) &&
!std::filesystem::create_directories(*EntryCache, ec)) {
LogMan::Msg::DFmt("Couldn't create EntryCache directory: '{}'", *EntryCache);
if (!std::filesystem::exists(EntryCache) &&
!std::filesystem::create_directories(EntryCache)) {
LogMan::Msg::D("Couldn't create EntryCache directory: '%s'", EntryCache.c_str());
}
}
void ShutdownPaths() {
CachePath.reset();
EntryCache.reset();
}
std::string GetCachePath() {
return *CachePath;
return CachePath;
}
std::string GetEntryCachePath() {
return *EntryCache;
return EntryCache;
}
}
-4
View File
@@ -3,10 +3,6 @@
namespace FEXCore::Paths {
void InitializePaths();
void ShutdownPaths();
const char *GetHomeDirectory();
std::string GetCachePath();
std::string GetEntryCachePath();
}
+27 -315
View File
@@ -1,6 +1,4 @@
#pragma once
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <cmath>
@@ -16,19 +14,9 @@ extern "C" {
struct X80SoftFloat {
#ifdef _M_X86_64
// Define this to push some operations to x87
// Only useful to see if precision loss is killing something
// #define DEBUG_X86_FLOAT
#ifdef DEBUG_X86_FLOAT
#define BIGFLOAT long double
#define BIGFLOATSIZE 10
#else
#define BIGFLOAT __float128
#define BIGFLOATSIZE 16
#endif
#elif defined(_M_ARM_64)
#define BIGFLOAT long double
#define BIGFLOATSIZE 16
#else
#error No 128bit float for this target!
#endif
@@ -57,183 +45,51 @@ struct X80SoftFloat {
// Ops
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
faddp;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_add(lhs, rhs);
#endif
}
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fsubp;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_sub(lhs, rhs);
#endif
}
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fmulp;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_mul(lhs, rhs);
#endif
}
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fdivp;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_div(lhs, rhs);
#endif
}
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fprem;
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
X80SoftFloat Rem = extF80_rem(lhs, rhs);
if (SignBit(Rem)) {
Rem = extF80_add(Rem, rhs);
}
else {
Rem.Sign = SignBit(lhs);
}
return Result;
#else
return extF80_rem(lhs, rhs);
#endif
return Rem;
}
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fprem1;
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_rem(lhs, rhs);
#endif
}
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
}
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
return extF80_roundToInt(lhs, RoundMode, false);
}
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fxtract;
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st", "st(1)");
return Result;
#else
X80SoftFloat Tmp = lhs;
Tmp.Exponent = 0x3FFF;
Tmp.Sign = lhs.Sign;
return Tmp;
#endif
}
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fxtract;
ffreep %%st(0);
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st", "st(1)");
return Result;
#else
int32_t TrueExp = lhs.Exponent - ExponentBias;
return i32_to_extF80(TrueExp);
#endif
}
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
@@ -243,211 +99,77 @@ struct X80SoftFloat {
}
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fscale; # st0 = st0 * 2^(rdint(st1))
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
X80SoftFloat Int = FRNDINT(rhs, softfloat_round_minMag);
WARN_ONCE("x87: Application used FSCALE which may have accuracy problems");
X80SoftFloat Int = FRNDINT(rhs);
BIGFLOAT Src2_d = Int;
Src2_d = exp2l(Src2_d);
X80SoftFloat Src2_X80 = Src2_d;
X80SoftFloat Result = extF80_mul(lhs, Src2_X80);
return Result;
#endif
}
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
f2xm1; # st0 = 2^st(0) - 1
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
WARN_ONCE("x87: Application used F2XM1 which may have accuracy problems");
BIGFLOAT Src1_d = lhs;
BIGFLOAT Result = exp2l(Src1_d);
Result -= 1.0;
return Result;
#endif
}
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st(1)
fldt %[lhs]; # st(0)
fyl2x; # st(1) * log2l(st(0))
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
WARN_ONCE("x87: Application used FYL2X which may have accuracy problems");
BIGFLOAT Src1_d = lhs;
BIGFLOAT Src2_d = rhs;
BIGFLOAT Tmp = Src2_d * log2l(Src1_d);
return Tmp;
#endif
}
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs];
fldt %[rhs];
fpatan;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
WARN_ONCE("x87: Application used FATAN which may have accuracy problems");
BIGFLOAT Src1_d = lhs;
BIGFLOAT Src2_d = rhs;
BIGFLOAT Tmp = atan2l(Src1_d, Src2_d);
return Tmp;
#endif
}
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fptan;
ffreep %%st(0);
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
WARN_ONCE("x87: Application used FTAN which may have accuracy problems");
BIGFLOAT Src_d = lhs;
Src_d = tanl(Src_d);
return Src_d;
#endif
}
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fsin;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
WARN_ONCE("x87: Application used FSIN which may have accuracy problems");
BIGFLOAT Src_d = lhs;
Src_d = sinl(Src_d);
return Src_d;
#endif
}
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fcos;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
WARN_ONCE("x87: Application used FCOS which may have accuracy problems");
BIGFLOAT Src_d = lhs;
Src_d = cosl(Src_d);
return Src_d;
#endif
}
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fsqrt;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
return extF80_sqrt(lhs);
#endif
}
operator float() const {
const float32_t Result = extF80_to_f32(*this);
return FEXCore::BitCast<float>(Result);
float32_t Result = extF80_to_f32(*this);
return *(float*)&Result;
}
operator double() const {
const float64_t Result = extF80_to_f64(*this);
return FEXCore::BitCast<double>(Result);
float64_t Result = extF80_to_f64(*this);
return *(double*)&Result;
}
operator BIGFLOAT() const {
#if BIGFLOATSIZE == 16
const float128_t Result = extF80_to_f128(*this);
return FEXCore::BitCast<BIGFLOAT>(Result);
#else
BIGFLOAT result{};
memcpy(&result, this, sizeof(result));
return result;
#endif
float128_t Result = extF80_to_f128(*this);
return *(BIGFLOAT*)&Result;
}
operator int16_t() const {
@@ -474,11 +196,11 @@ struct X80SoftFloat {
}
void operator=(const float rhs) {
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
*this = f32_to_extF80(*(float32_t*)&rhs);
}
void operator=(const double rhs) {
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
*this = f64_to_extF80(*(float64_t*)&rhs);
}
void operator=(const int16_t rhs) {
@@ -493,12 +215,6 @@ struct X80SoftFloat {
*this = ui64_to_extF80(rhs);
}
#if BIGFLOATSIZE == 10
void operator=(const long double rhs) {
memcpy(this, &rhs, sizeof(rhs));
}
#endif
operator void*() {
return reinterpret_cast<void*>(this);
}
@@ -510,19 +226,15 @@ struct X80SoftFloat {
}
X80SoftFloat(const float rhs) {
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
*this = f32_to_extF80(*(float32_t*)&rhs);
}
X80SoftFloat(const double rhs) {
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
*this = f64_to_extF80(*(float64_t*)&rhs);
}
X80SoftFloat(BIGFLOAT rhs) {
#if BIGFLOATSIZE == 16
*this = f128_to_extF80(FEXCore::BitCast<float128_t>(rhs));
#else
*this = FEXCore::BitCast<long double>(rhs);
#endif
*this = f128_to_extF80(*(float128_t*)&rhs);
}
X80SoftFloat(const int16_t rhs) {
-29
View File
@@ -1,29 +0,0 @@
#pragma once
#include <string>
namespace FEXCore::StringUtils {
// Trim the left side of the string of whitespace and new lines
[[maybe_unused]] static std::string LeftTrim(std::string String, std::string TrimTokens = " \t\n\r") {
size_t pos = std::string::npos;
if ((pos = String.find_first_not_of(TrimTokens)) != std::string::npos) {
String.erase(0, pos);
}
return String;
}
// Trim the right side of the string of whitespace and new lines
[[maybe_unused]] static std::string RightTrim(std::string String, std::string TrimTokens = " \t\n\r") {
size_t pos = std::string::npos;
if ((pos = String.find_last_not_of(TrimTokens)) != std::string::npos) {
String.erase(String.begin() + pos + 1, String.end());
}
return String;
}
// Trim both the left and right of the string of whitespace and new lines
[[maybe_unused]] static std::string Trim(std::string String, std::string TrimTokens = " \t\n\r") {
return RightTrim(LeftTrim(String, TrimTokens), TrimTokens);
}
}
+59 -431
View File
@@ -1,127 +1,43 @@
#include "Common/StringConv.h"
#include "Common/StringUtils.h"
#include "Common/Paths.h"
#include "Utils/FileLoading.h"
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Context/Context.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
#include <assert.h>
#include <cstdlib>
#include <filesystem>
#include <fstream>
#include <functional>
#include <pwd.h>
#include <map>
#include <memory>
#include <list>
#include <optional>
#include <stddef.h>
#include <stdint.h>
#include <string>
#include <string_view>
#include <sys/sysinfo.h>
#include <system_error>
#include <type_traits>
#include <unordered_map>
#include <utility>
#include <vector>
#include <tiny-json.h>
namespace FEXCore::Context {
struct Context;
}
#include <unistd.h>
namespace FEXCore::Config {
namespace DefaultValues {
#define P(x) x
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
#include <FEXCore/Config/ConfigValues.inl>
}
namespace JSON {
struct JsonAllocator {
jsonPool_t PoolObject;
std::unique_ptr<std::list<json_t>> json_objects;
};
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
json_t* PoolInit(jsonPool_t* Pool) {
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
alloc->json_objects = std::make_unique<std::list<json_t>>();
return &*alloc->json_objects->emplace(alloc->json_objects->end());
char const* FindUserHomeThroughUID() {
auto passwd = getpwuid(geteuid());
if (passwd) {
return passwd->pw_dir;
}
return nullptr;
}
json_t* PoolAlloc(jsonPool_t* Pool) {
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
return &*alloc->json_objects->emplace(alloc->json_objects->end());
}
const char *GetHomeDirectory() {
char const *HomeDir = getenv("HOME");
static void LoadJSonConfig(const std::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
std::vector<char> Data;
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
return;
// Try to get home directory from uid
if (!HomeDir) {
HomeDir = FindUserHomeThroughUID();
}
JsonAllocator Pool {
.PoolObject = {
.init = PoolInit,
.alloc = PoolAlloc,
},
};
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
if (!json) {
LogMan::Msg::EFmt("Couldn't create json");
return;
// try the PWD
if (!HomeDir) {
HomeDir = getenv("PWD");
}
json_t const* ConfigList = json_getProperty(json, "Config");
if (!ConfigList) {
// This is a non-error if the configuration file exists but no Config section
return;
// Still doesn't exit? You get local
if (!HomeDir) {
HomeDir = ".";
}
for (json_t const* ConfigItem = json_getChild(ConfigList);
ConfigItem != nullptr;
ConfigItem = json_getSibling(ConfigItem)) {
const char* ConfigName = json_getName(ConfigItem);
const char* ConfigString = json_getValue(ConfigItem);
if (!ConfigName) {
LogMan::Msg::EFmt("Couldn't get config name");
return;
}
if (!ConfigString) {
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
return;
}
Func(ConfigName, ConfigString);
}
}
}
std::string GetDataDirectory() {
std::string DataDir{};
char const *HomeDir = Paths::GetHomeDirectory();
char const *DataXDG = getenv("XDG_DATA_HOME");
char const *DataOverride = getenv("FEX_APP_DATA_LOCATION");
if (DataOverride) {
// Data override will override the complete directory
DataDir = DataOverride;
}
else {
DataDir = DataXDG ?: HomeDir;
DataDir += "/.fex-emu/";
}
return DataDir;
return HomeDir;
}
std::string GetConfigDirectory(bool Global) {
@@ -130,22 +46,15 @@ namespace JSON {
ConfigDir = GLOBAL_DATA_DIRECTORY;
}
else {
char const *HomeDir = Paths::GetHomeDirectory();
char const *HomeDir = GetHomeDirectory();
char const *ConfigXDG = getenv("XDG_CONFIG_HOME");
char const *ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
if (ConfigOverride) {
// Config override completely overrides the config directory
ConfigDir = ConfigOverride;
}
else {
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
ConfigDir += "/.fex-emu/";
}
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
ConfigDir += "/.fex-emu/";
// Ensure the folder structure is created for our configuration
std::error_code ec{};
if (!std::filesystem::exists(ConfigDir, ec) &&
!std::filesystem::create_directories(ConfigDir, ec)) {
if (!std::filesystem::exists(ConfigDir) &&
!std::filesystem::create_directories(ConfigDir)) {
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigDir.c_str());
// Let's go local in this case
return "./";
}
@@ -154,50 +63,35 @@ namespace JSON {
return ConfigDir;
}
std::string GetConfigFileLocation(bool Global) {
std::string ConfigFile{};
if (Global) {
ConfigFile = GetConfigDirectory(true) + "Config.json";
}
else {
const char *AppConfig = getenv("FEX_APP_CONFIG");
if (AppConfig) {
// App config environment variable overwrites only the config file
ConfigFile = AppConfig;
}
else {
ConfigFile = GetConfigDirectory(false) + "Config.json";
}
}
std::string GetConfigFileLocation() {
std::string ConfigFile = GetConfigDirectory(false) + "Config.json";
return ConfigFile;
}
std::string GetApplicationConfig(const std::string &Filename, bool Global) {
std::string GetApplicationConfig(std::string &Filename, bool Global) {
std::string ConfigFile = GetConfigDirectory(Global);
std::error_code ec{};
if (!Global &&
!std::filesystem::exists(ConfigFile, ec) &&
!std::filesystem::create_directories(ConfigFile, ec)) {
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
!std::filesystem::exists(ConfigFile) &&
!std::filesystem::create_directories(ConfigFile)) {
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigFile.c_str());
// Let's go local in this case
return "./" + Filename + ".json";
return "./";
}
ConfigFile += "AppConfig/";
// Attempt to create the local folder if it doesn't exist
if (!Global &&
!std::filesystem::exists(ConfigFile, ec) &&
!std::filesystem::create_directories(ConfigFile, ec)) {
// Let's go local in this case
return "./" + Filename + ".json";
}
ConfigFile += Filename + ".json";
ConfigFile += "AppConfig/" + Filename + ".json";
return ConfigFile;
}
std::string GetDataDirectory() {
std::string DataDir{};
char const *HomeDir = GetHomeDirectory();
char const *DataXDG = getenv("XDG_DATA_HOME");
DataDir = DataXDG ?: HomeDir;
DataDir += "/.fex-emu/";
return DataDir;
}
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
}
@@ -211,8 +105,7 @@ namespace JSON {
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
static FEXCore::Config::Layer *Meta{};
constexpr std::array<FEXCore::Config::LayerType, 7> LoadOrder = {
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
constexpr std::array<FEXCore::Config::LayerType, 6> LoadOrder = {
FEXCore::Config::LayerType::LAYER_MAIN,
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
@@ -298,8 +191,7 @@ namespace JSON {
void MetaLayer::MergeConfigMap(const LayerOptions &Options) {
// Insert this layer's options, overlaying previous options that exist here
for (auto &it : Options) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV ||
it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV) {
MergeEnvironmentVariables(it.first, it.second);
}
else {
@@ -327,7 +219,7 @@ namespace JSON {
}
}
std::string ExpandPath(std::string const &ContainerPrefix, std::string PathName) {
std::string ExpandPath(std::string PathName) {
if (PathName.empty()) {
return {};
}
@@ -348,67 +240,10 @@ namespace JSON {
Path = std::filesystem::absolute(Path);
// Only return if it exists
std::error_code ec{};
if (std::filesystem::exists(Path, ec)) {
if (std::filesystem::exists(Path)) {
return Path;
}
}
else {
// If the containerprefix and pathname isn't empty
// Then we check if the pathname exists in our current namespace
// If the path DOESN'T exist but DOES exist with the prefix applied
// then redirect to the prefix
//
// This might not be expected behaviour for some edge cases but since
// all paths aren't mounted inside the container, then it'll be fine
//
// Main catch case for this is the default thunk install folders
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
if (!ContainerPrefix.empty() && !PathName.empty()) {
if (!std::filesystem::exists(PathName)) {
auto ContainerPath = ContainerPrefix + PathName;
if (std::filesystem::exists(ContainerPath)) {
return ContainerPath;
}
}
}
}
return {};
}
std::string FindContainer() {
// We only support pressure-vessel at the moment
const static std::string ContainerManager = "/run/host/container-manager";
if (std::filesystem::exists(ContainerManager)) {
std::vector<char> Manager{};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
std::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
return ManagerStr;
}
}
return {};
}
std::string FindContainerPrefix() {
// We only support pressure-vessel at the moment
const static std::string ContainerManager = "/run/host/container-manager";
if (std::filesystem::exists(ContainerManager)) {
std::vector<char> Manager{};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
std::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
// We are running inside of pressure vessel
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
return "/run/host/";
}
}
}
return {};
}
@@ -416,9 +251,7 @@ namespace JSON {
Meta->Load();
// Do configuration option fix ups after everything is reloaded
{
// Always fix up the number of threads and create the configuration
// Otherwise the application could receive zero as the number of threads
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THREADS)) {
FEX_CONFIG_OPT(Cores, THREADS);
if (Cores == 0) {
// When the number of emulated CPU cores is zero then auto detect
@@ -426,38 +259,8 @@ namespace JSON {
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
// Sanitize Core option
FEX_CONFIG_OPT(Core, CORE);
#if (_M_X86_64)
constexpr uint32_t MaxCoreNumber = 2;
#else
constexpr uint32_t MaxCoreNumber = 1;
#endif
#ifdef INTERPRETER_ENABLED
constexpr uint32_t MinCoreNumber = 0;
#else
constexpr uint32_t MinCoreNumber = 1;
#endif
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
// Sanitize the core option by setting the core to the JIT if invalid
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, std::to_string(FEXCore::Config::CONFIG_IRJIT));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(Core, CORE);
if (CacheObjectCodeCompilation() && Core() == FEXCore::Config::CONFIG_INTERPRETER) {
// If running the interpreter then disable cache code compilation
FEXCore::Config::Erase(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION);
}
}
std::string ContainerPrefix { FindContainerPrefix() };
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
auto NewPath = ExpandPath(ContainerPrefix, PathName);
auto ExpandPathIfExists = [](FEXCore::Config::ConfigOption Config, std::string PathName) {
auto NewPath = ExpandPath(PathName);
if (!NewPath.empty()) {
FEXCore::Config::EraseSet(Config, NewPath);
}
@@ -465,7 +268,7 @@ namespace JSON {
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
FEX_CONFIG_OPT(PathName, ROOTFS);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
auto ExpandedString = ExpandPath(PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
@@ -473,8 +276,7 @@ namespace JSON {
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
std::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
std::error_code ec{};
if (std::filesystem::exists(NamedRootFS, ec)) {
if (std::filesystem::exists(NamedRootFS)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
}
}
@@ -489,19 +291,7 @@ namespace JSON {
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
}
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
std::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
std::error_code ec{};
if (std::filesystem::exists(NamedConfig, ec)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
}
}
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKCONFIG, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
@@ -532,15 +322,11 @@ namespace JSON {
return Meta->Get(Option);
}
void Set(ConfigOption Option, std::string_view Data) {
void Set(ConfigOption Option, std::string Data) {
Meta->Set(Option, Data);
}
void Erase(ConfigOption Option) {
Meta->Erase(Option);
}
void EraseSet(ConfigOption Option, std::string_view Data) {
void EraseSet(ConfigOption Option, std::string Data) {
Meta->EraseSet(Option, Data);
}
@@ -579,17 +365,6 @@ namespace JSON {
}
}
template<>
std::string Value<std::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
}
else {
return std::string(Default);
}
}
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
@@ -614,152 +389,5 @@ namespace JSON {
*List = **Value;
}
}
template void Value<std::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, std::list<std::string> *List);
// Application loaders
class MainLoader final : public FEXCore::Config::OptionMapper {
public:
explicit MainLoader(FEXCore::Config::LayerType Type);
explicit MainLoader(std::string ConfigFile);
void Load() override;
private:
std::string Config;
};
class AppLoader final : public FEXCore::Config::OptionMapper {
public:
explicit AppLoader(const std::string& Filename, bool Global);
void Load();
private:
std::string Config;
};
class EnvLoader final : public FEXCore::Config::Layer {
public:
explicit EnvLoader(char *const _envp[]);
void Load() override;
private:
char *const *envp;
};
static const std::map<std::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
#include <FEXCore/Config/ConfigValues.inl>
}};
static const std::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
#include <FEXCore/Config/ConfigValues.inl>
}};
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
: FEXCore::Config::Layer(Layer) {
}
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
auto it = ConfigLookup.find(ConfigName);
if (it != ConfigLookup.end()) {
Set(it->second, ConfigString);
}
}
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
: FEXCore::Config::OptionMapper(Type)
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
}
MainLoader::MainLoader(std::string ConfigFile)
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
, Config{std::move(ConfigFile)} {
}
void MainLoader::Load() {
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
MapNameToOption(Name, ConfigString);
});
}
AppLoader::AppLoader(const std::string& Filename, bool Global)
: FEXCore::Config::OptionMapper(Global ? FEXCore::Config::LayerType::LAYER_GLOBAL_APP : FEXCore::Config::LayerType::LAYER_LOCAL_APP) {
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
// Immediately load so we can reload the meta layer
Load();
}
void AppLoader::Load() {
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
MapNameToOption(Name, ConfigString);
});
}
EnvLoader::EnvLoader(char *const _envp[])
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
, envp {_envp} {
}
void EnvLoader::Load() {
std::unordered_map<std::string_view, std::string_view> EnvMap;
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
std::string_view Var(*pvar);
size_t pos = Var.rfind('=');
if (std::string::npos == pos)
continue;
std::string_view Key = Var.substr(0,pos);
std::string_view Value {Var.substr(pos+1)};
#define ENVLOADER
#include <FEXCore/Config/ConfigOptions.inl>
EnvMap[Key]=Value;
}
std::function GetVar = [=](const std::string_view id) -> std::optional<std::string_view> {
if (EnvMap.find(id) != EnvMap.end())
return EnvMap.at(id);
// If envp[] was empty, search using std::getenv()
const char* vs = std::getenv(id.data());
if (vs) {
return vs;
}
else {
return std::nullopt;
}
};
std::optional<std::string_view> Value;
for (auto &it : EnvConfigLookup) {
if ((Value = GetVar(it.first)).has_value()) {
Set(it.second, std::string(*Value));
}
}
}
std::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
}
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
if (File) {
return std::make_unique<FEXCore::Config::MainLoader>(*File);
}
else {
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
}
}
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, bool Global) {
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Global);
}
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
return std::make_unique<FEXCore::Config::EnvLoader>(_envp);
}
}
+232
View File
@@ -0,0 +1,232 @@
{
"Options": {
"CPU": {
"Core": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
"TextDefault": "irjit",
"ShortArg": "c",
"Choices": [ "irint", "irjit", "host" ],
"ArgumentHandler": "CoreHandler",
"Desc": [
"Which CPU core to use",
"host only exists on x86_64",
"[irint, irjit, host]"
]
},
"Multiblock": {
"Type": "bool",
"Default": "true",
"ShortArg": "m",
"Desc": [
"Controls multiblock code compilation"
]
},
"MaxInst": {
"Type": "int32",
"Default": "5000",
"ShortArg": "n",
"Desc": [
"Maximum number of instruction to store in a block"
]
},
"Threads": {
"Type": "uint32",
"Default": "1",
"ShortArg": "T",
"Desc": [
"Number of physical hardware threads to tell the process we have.",
"0 will auto detect."
]
}
},
"Emulation": {
"RootFS": {
"Type": "str",
"Default": "",
"ShortArg": "R",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
"\teg: ~/RootFS/Debian_x86_64",
"Or this can be a name of a rootfs",
"If the named rootfs exists in the FEX data folder then it will use that one",
"\teg: $HOME/.fex-emu/RootFS/<RootFS name>/",
"Or if you have XDG_DATA_HOME the config will search in that directory",
"\teg: $XDG_DATA_HOME/.fex-emu/RootFS/<RootFS name>/"
]
},
"ThunkHostLibs": {
"Type": "str",
"Default": "",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
]
},
"ThunkGuestLibs": {
"Type": "str",
"Default": "",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
},
"ThunkConfig": {
"Type": "str",
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks."
]
},
"Env": {
"Type": "strarray",
"Default": "",
"ShortArg": "E",
"Desc": [
"Adds an environment variable to the emulated environment."
]
}
},
"Debug": {
"SingleStep": {
"Type": "bool",
"Default": "false",
"ShortArg": "S",
"Desc": [
"Single stepping configuration."
]
},
"GdbServer": {
"Type": "bool",
"Default": "false",
"ShortArg": "G",
"Desc": [
"Enables the GDB server."
]
},
"DumpIR": {
"Type": "str",
"Default": "no",
"Desc": [
"Folder to dump the IR in to.",
"[no, stdout, stderr, <Folder>]"
]
},
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"ShortArg": "g",
"Desc": [
"When the test harness ends, print the GPR state."
]
},
"O0": {
"Type": "bool",
"Default": "false",
"ShortArg": "O0",
"Desc": [
"Disables optimizations passes for debugging."
]
}
},
"Logging": {
"SilentLog": {
"Type": "bool",
"Default": "true",
"ShortArg": "s",
"Desc": [
"Disables logging"
]
},
"OutputLog": {
"Type": "str",
"Default": "stdout",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
"[stdout, stderr, <Filename>]"
]
}
},
"Hacks": {
"SMCChecks": {
"Type": "uint8",
"Default": "FEXCore::Config::CONFIG_SMC_MMAN",
"TextDefault": "mman",
"ArgumentHandler": "SMCCheckHandler",
"Desc": [
"Checks code for modification before execution.",
"\tnone: No checks",
"\tmman: Invalidate on mmap, mprotect, munmap",
"\tfull: Validate code before every run (slow)"
]
},
"TSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"Controls TSO IR ops.",
"Highly likely to break any multithreaded application if disabled."
]
},
"ABILocalFlags": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around flags.",
"Assumes flags are not used across cals.",
"Hand-written assembly can violate this assumption."
]
},
"ABINoPF": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around parity flag calculation.",
"Removes the calculation of the parity flag from GPR instructions.",
"Assuming no uses rely on it"
]
}
},
"Misc": {
"AOTIRCapture": {
"Type": "bool",
"Default": "false",
"Desc": [
"Captures IR and generates an AOT IR cache.",
"Captures both the loaded executable and libraries it loads."
]
},
"AOTIRLoad": {
"Type": "bool",
"Default": "false",
"Desc": [
"Loads an AOT IR cache for the loaded executable."
]
}
}
},
"UnnamedOptions": {
"Misc": {
"IS_INTERPRETER": {
"Type": "bool",
"Default": "false"
},
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false"
},
"APP_FILENAME": {
"Type": "str",
"Default": ""
},
"IS64BIT_MODE": {
"Type": "bool",
"Default": "false"
}
}
}
}
-403
View File
@@ -1,403 +0,0 @@
{
"Options": {
"CPU": {
"Core": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
"TextDefault": "irjit",
"ShortArg": "c",
"Choices": [ "irint", "irjit", "host" ],
"ArgumentHandler": "CoreHandler",
"Desc": [
"Which CPU core to use",
"host only exists on x86_64",
"[irint, irjit, host]"
]
},
"Multiblock": {
"Type": "bool",
"Default": "true",
"ShortArg": "m",
"Desc": [
"Controls multiblock code compilation"
]
},
"MaxInst": {
"Type": "int32",
"Default": "5000",
"ShortArg": "n",
"Desc": [
"Maximum number of instruction to store in a block"
]
},
"Threads": {
"Type": "uint32",
"Default": "0",
"ShortArg": "T",
"Desc": [
"Number of physical hardware threads to tell the process we have.",
"0 will auto detect."
]
},
"CacheObjectCodeCompilation": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
"TextDefault": "none",
"Choices": [ "none", "read", "readwrite" ],
"ArgumentHandler": "CacheObjectCodeHandler",
"Desc": [
"Cache JIT object code to drive.",
"Allows JIT code to be shared between applications"
]
},
"EnableAVX": {
"Type": "bool",
"Default": "true",
"Desc": [
"Determines whether or not we use the expanded register file for AVX or not"
]
}
},
"Emulation": {
"RootFS": {
"Type": "str",
"Default": "",
"ShortArg": "R",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
"\teg: ~/RootFS/Debian_x86_64",
"Or this can be a name of a rootfs",
"If the named rootfs exists in the FEX data folder then it will use that one",
"\teg: $HOME/.fex-emu/RootFS/<RootFS name>/",
"Or if you have XDG_DATA_HOME the config will search in that directory",
"\teg: $XDG_DATA_HOME/.fex-emu/RootFS/<RootFS name>/"
]
},
"ThunkHostLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks/",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
]
},
"ThunkGuestLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks/",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
},
"ThunkGuestLibs32": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks_32/",
"Desc": [
"Folder to find the 32-bit guest-side thunking libraries."
]
},
"ThunkConfig": {
"Type": "str",
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks.",
"This can be a filesystem path",
"\teg: ~/MyThunkConfig.json",
"Or this can be a named of a Thunk config file",
"If the named config file exists in the FEX data folder folder the it will use that one",
"\teg: $HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>",
"Or if you have XDG_DATA_HOME the config will search in that directory",
"\teg: $XDG_DATA_HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>"
]
},
"Env": {
"Type": "strarray",
"Default": "",
"ShortArg": "E",
"Desc": [
"Adds an environment variable to the emulated environment."
]
},
"HostEnv": {
"Type": "strarray",
"Default": "",
"ShortArg": "H",
"Desc": [
"Adds an environment variable to the host environment.",
"This can be useful for setting environment variables that thunks can pick up.",
"Typically isn't necessary since the guest libc isn't thunked. But is possible."
]
},
"AdditionalArguments": {
"Type": "strarray",
"Default": "",
"Desc": [
"Allows the user to pass additional arguments to the application"
]
}
},
"Debug": {
"SingleStep": {
"Type": "bool",
"Default": "false",
"ShortArg": "S",
"Desc": [
"Single stepping configuration."
]
},
"GdbServer": {
"Type": "bool",
"Default": "false",
"ShortArg": "G",
"Desc": [
"Enables the GDB server."
]
},
"DumpIR": {
"Type": "str",
"Default": "no",
"Desc": [
"Folder to dump the IR in to.",
"[no, stdout, stderr, <Folder>]"
]
},
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"ShortArg": "g",
"Desc": [
"When the test harness ends, print the GPR state."
]
},
"O0": {
"Type": "bool",
"Default": "false",
"ShortArg": "O0",
"Desc": [
"Disables optimizations passes for debugging."
]
},
"SRA": {
"Type": "bool",
"Default": "true",
"Desc": [
"Set to false to disable Static Register Allocation"
]
},
"Force32BitAllocator": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces use of the 32-bit allocator on 32-bit applications",
"Used to work around ulimit problems of CI runner",
"Potentially useful for debugging memory problems",
"32-bit allocator is always used if your host kernel is older than 4.17"
]
},
"GlobalJITNaming": {
"Type": "bool",
"Default": "false",
"Desc": [
"Uses JITSymbols to name all JIT state as one symbol",
"Useful for querying how much time is spent inside of the JIT",
"Profiling tools will show JIT time as FEXJIT"
]
},
"LibraryJITNaming": {
"Type": "bool",
"Default": "false",
"Desc": [
"Uses JITSymbols to name JIT symbols grouped by library",
"Useful for querying how much time is spent in each guest library",
"Can be used to help guide thunk generation"
]
},
"BlockJITNaming": {
"Type": "bool",
"Default": "false",
"Desc": [
"Uses JITSymbols to name JIT symbols",
"Useful for determining hot blocks of code",
"Has some file writing overhead per JIT block"
]
},
"GDBSymbols": {
"Type": "bool",
"Default": "false",
"Desc": [
"Integrates with GDB using the JIT interface.",
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
"Also needs x86_64-linux-gnu-objdump in PATH.",
"Can be very slow."
]
}
},
"Logging": {
"SilentLog": {
"Type": "bool",
"Default": "true",
"ShortArg": "s",
"Desc": [
"Disables logging"
]
},
"OutputLog": {
"Type": "str",
"Default": "server",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
"[stdout, stderr, server, <Filename>]"
]
}
},
"Hacks": {
"SMCChecks": {
"Type": "uint8",
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
"TextDefault": "mtrack",
"ArgumentHandler": "SMCCheckHandler",
"Desc": [
"Checks code for modification before execution.",
"\tnone: No checks",
"\tmtrack: Page tracking based invalidation",
"\tfull: Validate code before every run (slow)",
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
]
},
"TSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"Controls TSO IR ops.",
"Highly likely to break any multithreaded application if disabled."
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
"Desc": [
"Automatically enables TSO when shared memory is used.",
"Should work without issues in most cases."
]
},
"X87ReducedPrecision": {
"Type": "bool",
"Default": "false",
"Desc": [
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
]
},
"ABILocalFlags": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around flags.",
"Assumes flags are not used across cals.",
"Hand-written assembly can violate this assumption."
]
},
"ABINoPF": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around parity flag calculation.",
"Removes the calculation of the parity flag from GPR instructions.",
"Assuming no uses rely on it"
]
},
"ParanoidTSO": {
"Type": "bool",
"Default": "false",
"Desc": [
"Makes TSO operations even more strict.",
"Forces vector loadstores to also become atomic."
]
},
"StallProcess": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces a process to stall out on initialization",
"Useful for a process that keeps restarting and doesn't work"
]
},
"x86dec_SynchronizeRIPOnAllBlocks": {
"Type": "bool",
"Default": "false",
"Desc": [
"An application that uses try-catch or longjump extensively needs the ability to do context aware state flushing",
"In the case of FEX's block-linking, it won't always ensure that RIP is synchronized.",
"If an exception occurs and RIP isn't synchronized, then FEX's exception stack restore may not long jump as expected",
"Can be useful for Wine applications that rely on stack unwinding"
]
}
},
"Misc": {
"AOTIRCapture": {
"Type": "bool",
"Default": "false",
"Desc": [
"Captures IR and generates an AOT IR cache.",
"Captures both the loaded executable and libraries it loads."
]
},
"AOTIRGenerate": {
"Type": "bool",
"Default": "false",
"Desc": [
"Scans file for executable code and generates an AOT IR cache.",
"Does not run the executable."
]
},
"AOTIRLoad": {
"Type": "bool",
"Default": "false",
"Desc": [
"Loads an AOT IR cache for the loaded executable."
]
},
"ServerSocketPath": {
"Type": "str",
"Default": "",
"Desc": [
"Override for a FEXServer socket path. Only useful for chroots."
]
}
}
},
"UnnamedOptions": {
"Misc": {
"IS_INTERPRETER": {
"Type": "bool",
"Default": "false"
},
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false"
},
"APP_FILENAME": {
"Type": "str",
"Default": ""
},
"APP_CONFIG_NAME": {
"Type": "str",
"Default": "",
"Desc": [
"This is the application config name that has been loaded.",
"This differs from APP_FILENAME in two ways",
"Where APP_FILENAME always points to the executable path that FEX-Emu is executing.",
"This matches what is used to load the AppLayer configuration name.",
"When running through a compatibility layer like wine, this will only be the exe name, instead of wine full path."
]
},
"IS64BIT_MODE": {
"Type": "bool",
"Default": "false"
}
}
}
}
+28 -77
View File
@@ -2,20 +2,10 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/Core.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Core/SignalDelegator.h>
#include "FEXCore/Debug/InternalThreadState.h"
#include <string.h>
#include <utility>
namespace FEXCore::HLE {
class SyscallVisitor;
}
#include <FEXCore/Debug/X86Tables.h>
namespace FEXCore::Context {
void InitializeStaticTables(OperatingMode Mode) {
@@ -24,10 +14,6 @@ namespace FEXCore::Context {
IR::InstallOpcodeHandlers(Mode);
}
void ShutdownStaticTables() {
FEXCore::Paths::ShutdownPaths();
}
FEXCore::Context::Context *CreateNewContext() {
return new FEXCore::Context::Context{};
}
@@ -43,15 +29,16 @@ namespace FEXCore::Context {
delete CTX;
}
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, uint64_t InitialRIP, uint64_t StackPointer) {
return CTX->InitCore(InitialRIP, StackPointer);
bool InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
return CTX->InitCore(Loader);
}
void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler) {
CTX->CustomExitHandler = std::move(handler);
void SetExitHandler(FEXCore::Context::Context *CTX,
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> handler) {
CTX->CustomExitHandler = handler;
}
ExitHandler GetExitHandler(const FEXCore::Context::Context *CTX) {
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> GetExitHandler(FEXCore::Context::Context *CTX) {
return CTX->CustomExitHandler;
}
@@ -63,31 +50,28 @@ namespace FEXCore::Context {
CTX->Step();
}
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
Thread->CTX->CompileBlock(Thread->CurrentFrame, GuestRIP);
}
FEXCore::Context::ExitReason RunUntilExit(FEXCore::Context::Context *CTX) {
return CTX->RunUntilExit();
}
int GetProgramStatus(const FEXCore::Context::Context *CTX) {
int GetProgramStatus(FEXCore::Context::Context *CTX) {
return CTX->GetProgramStatus();
}
FEXCore::Context::ExitReason GetExitReason(const FEXCore::Context::Context *CTX) {
FEXCore::Context::ExitReason GetExitReason(FEXCore::Context::Context *CTX) {
return CTX->ParentThread->ExitReason;
}
bool IsDone(const FEXCore::Context::Context *CTX) {
bool IsDone(FEXCore::Context::Context *CTX) {
return CTX->IsPaused();
}
void GetCPUState(const FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
void GetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
memcpy(State, CTX->ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
}
void SetCPUState(FEXCore::Context::Context *CTX, const FEXCore::Core::CPUState *State) {
void SetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
memcpy(CTX->ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
}
@@ -110,30 +94,22 @@ namespace FEXCore::Context {
void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t Syscall, [[maybe_unused]] FEXCore::HLE::SyscallVisitor *Visitor) {
}
HostFeatures GetHostFeatures(const FEXCore::Context::Context *CTX) {
return CTX->HostFeatures;
void HandleCallback(FEXCore::Context::Context *CTX, uint64_t RIP) {
CTX->HandleCallback(RIP);
}
void HandleCallback(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
CTX->HandleCallback(Thread, RIP);
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
CTX->RegisterHostSignalHandler(Signal, Func);
}
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
CTX->RegisterHostSignalHandler(Signal, std::move(Func), Required);
}
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
CTX->RegisterFrontendHostSignalHandler(Signal, std::move(Func), Required);
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
CTX->RegisterFrontendHostSignalHandler(Signal, Func);
}
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
return CTX->CreateThread(NewThreadState, ParentTID);
}
void ExecutionThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
return CTX->ExecutionThread(Thread);
}
void InitializeThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
return CTX->InitializeThread(Thread);
}
@@ -153,57 +129,32 @@ namespace FEXCore::Context {
void CleanupAfterFork(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
CTX->CleanupAfterFork(Thread);
}
void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation) {
CTX->SignalDelegation = SignalDelegation;
}
void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler) {
CTX->SyscallHandler = Handler;
CTX->SourcecodeResolver = Handler->GetSourcecodeResolver();
}
FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf) {
return CTX->CPUID.RunFunction(Function, Leaf);
}
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf, uint32_t CPU) {
return CTX->CPUID.RunFunctionName(Function, Leaf, CPU);
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::istream>(const std::string&)> CacheReader) {
CTX->AOTIRLoader = CacheReader;
}
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader) {
CTX->SetAOTIRLoader(CacheReader);
bool WriteAOTIR(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
return CTX->WriteAOTIRCache(CacheWriter);
}
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
CTX->SetAOTIRWriter(CacheWriter);
void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name) {
return CTX->AddNamedRegion(Base, Length, Offset, Name);
}
void SetAOTIRRenamer(FEXCore::Context::Context *CTX, std::function<void(const std::string&)> CacheRenamer) {
CTX->SetAOTIRRenamer(CacheRenamer);
}
void FinalizeAOTIRCache(FEXCore::Context::Context *CTX) {
CTX->FinalizeAOTIRCache();
}
void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
CTX->WriteFilesWithCode(Writer);
}
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(FEXCore::Context::Context *CTX, const std::string &Name) {
return CTX->LoadAOTIRCacheEntry(Name);
}
void UnloadAOTIRCacheEntry(FEXCore::Context::Context *CTX, IR::AOTIRCacheEntry *Entry) {
return CTX->UnloadAOTIRCacheEntry(Entry);
}
CustomIRResult AddCustomIREntrypoint(FEXCore::Context::Context *CTX, uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
return CTX->AddCustomIREntrypoint(Entrypoint, Handler, Creator, Data);
}
void AppendThunkDefinitions(FEXCore::Context::Context *CTX, std::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
CTX->AppendThunkDefinitions(Definitions);
void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length) {
return CTX->RemoveNamedRegion(Base, Length);
}
namespace Debug {
+76 -214
View File
@@ -1,62 +1,44 @@
#pragma once
#include "Common/JitSymbols.h"
#include "FEXHeaderUtils/ScopedSignalMask.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/HostFeatures.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/AOTIR.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Utils/Event.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <stdint.h>
#include <atomic>
#include <condition_variable>
#include <functional>
#include <istream>
#include <map>
#include <memory>
#include <mutex>
#include <shared_mutex>
#include <stddef.h>
#include <string>
#include <optional>
#include <ostream>
#include <set>
#include <unordered_map>
#include <queue>
#include <vector>
namespace FEXCore {
class CodeLoader;
class ThunkHandler;
class BlockSamplingData;
class GdbServer;
namespace CodeSerialize {
class CodeObjectSerializeService;
}
class SiganlDelegator;
namespace CPU {
class Arm64JITCore;
class X86JITCore;
class InterpreterCore;
class Dispatcher;
}
namespace HLE {
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
}
}
namespace FEXCore::IR {
class RegisterAllocationPass;
class RegisterAllocationData;
class IRListView;
namespace Validation {
@@ -79,7 +61,6 @@ namespace FEXCore::Context {
friend class FEXCore::CPU::X86JITCore;
#endif
friend class FEXCore::CPU::InterpreterCore;
friend class FEXCore::IR::Validation::IRValidation;
struct {
@@ -94,38 +75,28 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(ABINoPF, ABINOPF);
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(Core, CORE);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
FEX_CONFIG_OPT(DumpIR, DUMPIR);
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(x86dec_SynchronizeRIPOnAllBlocks, X86DEC_SYNCHRONIZERIPONALLBLOCKS);
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
} Config;
using IntCallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
IntCallbackReturn InterpreterCallbackReturn;
FEXCore::HostFeatures HostFeatures;
std::mutex ThreadCreationMutex;
uint64_t ThreadID{};
FEXCore::Core::InternalThreadState* ParentThread;
std::vector<FEXCore::Core::InternalThreadState*> Threads;
std::atomic_bool CoreShuttingDown{false};
bool NeedToCheckXID{true};
std::mutex IdleWaitMutex;
std::condition_variable IdleWaitCV;
@@ -134,16 +105,34 @@ namespace FEXCore::Context {
Event PauseWait;
bool Running{};
std::shared_mutex CodeInvalidationMutex;
FEXCore::CPUIDEmu CPUID;
FEXCore::HLE::SyscallHandler *SyscallHandler{};
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CustomCPUFactoryType CustomCPUFactory;
FEXCore::Context::ExitHandler CustomExitHandler;
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> CustomExitHandler;
struct AOTIRCacheEntry {
uint64_t start;
uint64_t len;
uint64_t crc;
IR::IRListView *IR;
IR::RegisterAllocationData *RAData;
};
std::function<std::unique_ptr<std::istream>(const std::string&)> AOTIRLoader;
std::unordered_map<std::string, std::map<uint64_t, AOTIRCacheEntry>> AOTIRCache;
struct AddrToFileEntry {
uint64_t Start;
uint64_t Len;
uint64_t Offset;
std::string fileid;
void *CachedFileEntry;
};
std::map<uint64_t, AddrToFileEntry> AddrToFile;
#ifdef BLOCKSTATS
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
@@ -155,9 +144,9 @@ namespace FEXCore::Context {
Context();
~Context();
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer);
bool InitCore(FEXCore::CodeLoader *Loader);
FEXCore::Context::ExitReason RunUntilExit();
int GetProgramStatus() const;
int GetProgramStatus();
bool IsPaused() const { return !Running; }
void Pause();
void Run();
@@ -168,40 +157,20 @@ namespace FEXCore::Context {
void StopThread(FEXCore::Core::InternalThreadState *Thread);
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
bool GetGdbServerStatus() { return (bool)DebugServer; }
void StartGdbServer();
void StopGdbServer();
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
void HandleCallback(uint64_t RIP);
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
static void RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
FHU::ScopedSignalMaskWithSharedLock lk(Frame->Thread->CTX->CodeInvalidationMutex);
return Fn(Frame, record);
// Wrapper which takes CpuStateFrame instead of InternalThreadState
static void RemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
RemoveCodeEntry(Frame->Thread, GuestRIP);
}
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
// Must be called from owning thread
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
// returns false if a handler was already registered
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data);
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
// Debugger interface
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
uint64_t GetThreadCount() const;
@@ -209,172 +178,65 @@ namespace FEXCore::Context {
bool GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data);
bool FindHostCodeForRIP(uint64_t RIP, uint8_t **Code);
struct GenerateIRResult {
FEXCore::IR::IRListView* IRList;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
uint64_t TotalInstructions;
uint64_t TotalInstructionsLength;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
// XXX:
// bool FindIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir);
// void SetIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir);
void LoadEntryList();
struct CompileCodeResult {
void* CompiledCode;
FEXCore::IR::IRListView* IRData;
FEXCore::Core::DebugData* DebugData;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
bool GeneratedIR;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
std::tuple<FEXCore::IR::IRListView *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t, uint64_t, uint64_t> GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
std::tuple<void *, FEXCore::IR::IRListView *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool, uint64_t, uint64_t> CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
// same as CompileBlock, but aborts on failure
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
bool LoadAOTIRCache(std::istream &stream);
bool WriteAOTIRCache(std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
// Used for thread creation from syscalls
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param NewThreadState The initial thread state to setup for our state
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* OS thread Creation:
* - Thread = CreateThread(NewState, PPID);
* - InitializeThread(Thread);
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
/**
* @brief Initializes TID, PID and TLS data for a thread
*
* @param Thread The internal FEX thread state object
*/
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
*
* @param Thread The internal FEX thread state object
*
* The OS thread will wait until RunThread is executed
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
void InitializeThread(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Starts the OS thread object to start executing guest code
*
* @param Thread The internal FEX thread state object
*/
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
void RunThread(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Destroys this FEX thread object and stops tracking it internally
*
* @param Thread The internal FEX thread state object
*/
void DestroyThread(FEXCore::Core::InternalThreadState *Thread);
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
void CleanupAfterFork(FEXCore::Core::InternalThreadState *ExceptForThread);
std::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
std::vector<FEXCore::Core::InternalThreadState*> *const GetThreads() { return &Threads; }
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const std::string &filename);
void UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry);
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
#if ENABLE_JITSYMBOLS
FEXCore::JITSymbols Symbols;
#endif
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread);
void FinalizeAOTIRCache() {
IRCaptureCache.FinalizeAOTIRCache();
}
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void SetAOTIRLoader(std::function<int(const std::string&)> CacheReader) {
IRCaptureCache.SetAOTIRLoader(CacheReader);
}
void SetAOTIRWriter(std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
IRCaptureCache.SetAOTIRWriter(CacheWriter);
}
void SetAOTIRRenamer(std::function<void(const std::string&)> CacheRenamer) {
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
}
void AppendThunkDefinitions(std::vector<FEXCore::IR::ThunkDefinition> const& Definitions);
FEXCore::Utils::PooledAllocatorMMap OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorMMap FrontendAllocator;
void MarkMemoryShared();
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
protected:
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
private:
/**
* @brief Does some final thread initialization
*
* @param Thread The internal FEX thread state object
*
* InitCore and CreateThread both call this to finish up thread object initialization
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Initializes the JIT compilers for the thread
*
* @param State The internal FEX thread state object
*
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
void WaitForIdleWithTimeout();
void NotifyPause();
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length);
FEXCore::CodeLoader *LocalLoader{};
// Entry Cache
std::optional<std::string> GetFilenameHash(std::string const &Filename) const;
void AddThreadRIPsToEntryList(FEXCore::Core::InternalThreadState *Thread);
void SaveEntryList();
std::set<uint64_t> EntryList;
std::vector<uint64_t> InitLocations;
uint64_t StartingRIP;
std::mutex ExitMutex;
std::unique_ptr<GdbServer> DebugServer;
IR::AOTIRCaptureCache IRCaptureCache;
std::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
bool StartPaused = false;
bool IsMemoryShared = false;
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
std::shared_mutex CustomIRMutex;
std::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
FEXCore::CPU::DispatcherConfig DispatcherConfig;
};
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
File diff suppressed because it is too large. Load diff
@@ -12,58 +12,6 @@ namespace FEXCore::ArchHelpers::Arm64 {
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
constexpr uint32_t RCPC2_MASK = 0x3F'E0'0C'00;
constexpr uint32_t LDAPUR_INST = 0x19'40'00'00;
constexpr uint32_t STLUR_INST = 0x19'00'00'00;
constexpr uint32_t LDAXP_MASK = 0xBF'FF'80'00;
constexpr uint32_t LDAXP_INST = 0x88'7F'80'00;
constexpr uint32_t STLXP_MASK = 0xBF'E0'80'00;
constexpr uint32_t STLXP_INST = 0x88'20'80'00;
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
constexpr uint32_t CBNZ_MASK = 0x7F'00'00'00;
constexpr uint32_t CBNZ_INST = 0x35'00'00'00;
constexpr uint32_t ALU_OP_MASK = 0x7F'20'00'00;
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
constexpr uint32_t ADD_SHIFT_INST = 0x0B'20'00'00;
constexpr uint32_t SUB_SHIFT_INST = 0x4B'20'00'00;
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
constexpr uint32_t CMP_SHIFT_INST = 0x6B'20'00'00;
constexpr uint32_t AND_INST = 0x0A'00'00'00;
constexpr uint32_t BIC_INST = 0x0A'20'00'00;
constexpr uint32_t OR_INST = 0x2A'00'00'00;
constexpr uint32_t ORN_INST = 0x2A'20'00'00;
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
constexpr uint32_t EON_INST = 0x4A'20'00'00;
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
constexpr uint32_t CLREX_MASK = 0xFF'FF'F0'FF;
constexpr uint32_t CLREX_INST = 0xD5'03'30'5F;
enum ExclusiveAtomicPairType {
TYPE_SWAP,
TYPE_ADD,
TYPE_SUB,
TYPE_AND,
TYPE_BIC,
TYPE_OR,
TYPE_ORN,
TYPE_EOR,
TYPE_EON,
TYPE_NEG, // This is just a sub with zero. Need to know the differences
};
// Load ops are 4 bits
// Acquire and release bits are independent on the instruction
constexpr uint32_t ATOMIC_ADD_OP = 0b0000;
@@ -76,34 +24,7 @@ namespace FEXCore::ArchHelpers::Arm64 {
constexpr uint32_t ATOMIC_UMIN_OP = 0b0111;
constexpr uint32_t ATOMIC_SWAP_OP = 0b1000;
constexpr uint32_t REGISTER_MASK = 0b11111;
constexpr uint32_t RD_OFFSET = 0;
constexpr uint32_t RN_OFFSET = 5;
constexpr uint32_t RM_OFFSET = 16;
constexpr uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
0b1011'0000'0000; // Inner shareable all
inline uint32_t GetRdReg(uint32_t Instr) {
return (Instr >> RD_OFFSET) & REGISTER_MASK;
}
inline uint32_t GetRnReg(uint32_t Instr) {
return (Instr >> RN_OFFSET) & REGISTER_MASK;
}
inline uint32_t GetRmReg(uint32_t Instr) {
return (Instr >> RM_OFFSET) & REGISTER_MASK;
}
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr);
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info);
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr);
uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr);
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr);
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr);
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr);
[[nodiscard]] bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext);
}
@@ -1,122 +1,50 @@
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Context/Context.h"
#include "Interface/HLE/Thunks/Thunks.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Core/CoreState.h>
#include <aarch64/cpu-aarch64.h>
#include <aarch64/instructions-aarch64.h>
#include <cpu-features.h>
#include <utils-vixl.h>
#include <array>
#include <tuple>
#include <utility>
#include "aarch64/cpu-aarch64.h"
namespace FEXCore::CPU {
#define STATE x28
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
: vixl::aarch64::Assembler(size ? (byte*)FEXCore::Allocator::mmap(nullptr, size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0) : reinterpret_cast<byte*>(~0ULL),
size,
vixl::aarch64::PositionDependentCode)
, EmitterCTX {ctx} {
Arm64Emitter::Arm64Emitter(size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
CPU.SetUp();
#ifdef VIXL_SIMULATOR
auto Features = vixl::CPUFeatures::All();
#else
auto Features = vixl::CPUFeatures::InferFromOS();
if (ctx->HostFeatures.SupportsAtomics) {
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
// RCPC is bugged on Snapdragon 865
// Causes glibc cond16 test to immediately throw assert
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
if (SupportsAtomics) {
// Hypervisor can hide this on the c630?
Features.Combine(vixl::CPUFeatures::Feature::kLORegions);
}
#endif
SetCPUFeatures(Features);
}
Arm64Emitter::~Arm64Emitter() {
auto CodeBuffer = GetBuffer();
if (CodeBuffer->GetCapacity()) {
FEXCore::Allocator::munmap(CodeBuffer->GetStartAddress<void*>(), CodeBuffer->GetCapacity());
if (!SupportsAtomics) {
WARN_ONCE("Host CPU doesn't support atomics. Expect bad performance");
}
}
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad) {
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) {
bool Is64Bit = Reg.IsX();
int Segments = Is64Bit ? 4 : 2;
if (Is64Bit && ((~Constant)>> 16) == 0) {
movn(Reg, (~Constant) & 0xFFFF);
if (NOPPad) {
nop(); nop(); nop();
}
return;
}
int NumMoves = 1;
int RequiredMoveSegments{};
// Count the number of move segments
// We only want to use ADRP+ADD if we have more than 1 segment
for (size_t i = 0; i < Segments; ++i) {
movz(Reg, (Constant) & 0xFFFF, 0);
for (int i = 1; i < Segments; ++i) {
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
if (Part != 0) {
++RequiredMoveSegments;
}
}
// ADRP+ADD is specifically optimized in hardware
// Check if we can use this
auto PC = GetCursorAddress<uint64_t>();
// PC aligned to page
uint64_t AlignedPC = PC & ~0xFFFULL;
// Offset from aligned PC
int64_t AlignedOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(AlignedPC);
// If the aligned offset is within the 4GB window then we can use ADRP+ADD
// and the number of move segments more than 1
if (RequiredMoveSegments > 1 && vixl::IsInt32(AlignedOffset)) {
// If this is 4k page aligned then we only need ADRP
if ((AlignedOffset & 0xFFF) == 0) {
adrp(Reg, AlignedOffset >> 12);
}
else {
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
// 21-bit signed integer here
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
if (vixl::IsInt21(SmallOffset)) {
adr(Reg, SmallOffset);
}
else {
// Need to use ADRP + ADD
adrp(Reg, AlignedOffset >> 12);
add(Reg, Reg, Constant & 0xFFF);
NumMoves = 2;
}
}
}
else {
movz(Reg, (Constant) & 0xFFFF, 0);
for (int i = 1; i < Segments; ++i) {
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
if (Part) {
movk(Reg, Part, i * 16);
++NumMoves;
}
}
}
if (NOPPad) {
for (int i = NumMoves; i < Segments; ++i) {
nop();
if (Part) {
movk(Reg, Part, i * 16);
}
}
}
@@ -125,7 +53,7 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
// We need to save pairs of registers
// We save r19-r30
MemOperand PairOffset(sp, -16, PreIndex);
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
{x19, x20},
{x21, x22},
{x23, x24},
@@ -188,7 +116,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
}
MemOperand PairOffset(sp, 16, PostIndex);
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
{x29, x30},
{x27, x28},
{x25, x26},
@@ -203,135 +131,41 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
}
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
if (StaticRegisterAllocation()) {
for (size_t i = 0; i < SRA64.size(); i+=2) {
auto Reg1 = SRA64[i];
auto Reg2 = SRA64[i+1];
if (((1U << Reg1.GetCode()) & GPRSpillMask) &&
((1U << Reg2.GetCode()) & GPRSpillMask)) {
stp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg1.GetCode()) & GPRSpillMask)) {
str(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg2.GetCode()) & GPRSpillMask)) {
str(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
}
}
void Arm64Emitter::SpillStaticRegs() {
for (size_t i = 0; i < SRA64.size(); i+=2) {
stp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
if (FPRs) {
if (EmitterCTX->HostFeatures.SupportsAVX) {
for (size_t i = 0; i < SRAFPR.size(); i++) {
const auto Reg = SRAFPR[i];
if (((1U << Reg.GetCode()) & FPRSpillMask) != 0) {
mov(TMP4, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
st1b(Reg.Z().VnB(), PRED_TMP_32B, SVEMemOperand(STATE, TMP4));
}
}
} else {
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
const auto Reg1 = SRAFPR[i];
const auto Reg2 = SRAFPR[i + 1];
if (((1U << Reg1.GetCode()) & FPRSpillMask) &&
((1U << Reg2.GetCode()) & FPRSpillMask)) {
stp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
}
else if (((1U << Reg1.GetCode()) & FPRSpillMask)) {
str(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
}
else if (((1U << Reg2.GetCode()) & FPRSpillMask)) {
str(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
}
}
}
}
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
stp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
}
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
if (StaticRegisterAllocation()) {
for (size_t i = 0; i < SRA64.size(); i+=2) {
auto Reg1 = SRA64[i];
auto Reg2 = SRA64[i+1];
if (((1U << Reg1.GetCode()) & GPRFillMask) &&
((1U << Reg2.GetCode()) & GPRFillMask)) {
ldp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg1.GetCode()) & GPRFillMask)) {
ldr(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg2.GetCode()) & GPRFillMask)) {
ldr(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
}
}
void Arm64Emitter::FillStaticRegs() {
for (size_t i = 0; i < SRA64.size(); i+=2) {
ldp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
if (FPRs) {
if (EmitterCTX->HostFeatures.SupportsAVX) {
// Set up predicate registers.
// We don't bother spilling these in SpillStaticRegs,
// since all that matters is we restore them on a fill.
// It's not a concern if they get trounced by something else.
ptrue(PRED_TMP_16B.VnB(), SVE_VL16);
ptrue(PRED_TMP_32B.VnB(), SVE_VL32);
for (size_t i = 0; i < SRAFPR.size(); i++) {
const auto Reg = SRAFPR[i];
if (((1U << Reg.GetCode()) & FPRFillMask) != 0) {
mov(TMP4, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
ld1b(Reg.Z().VnB(), PRED_TMP_32B.Zeroing(), SVEMemOperand(STATE, TMP4));
}
}
} else {
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
const auto Reg1 = SRAFPR[i];
const auto Reg2 = SRAFPR[i + 1];
if (((1U << Reg1.GetCode()) & FPRFillMask) &&
((1U << Reg2.GetCode()) & FPRFillMask)) {
ldp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
}
else if (((1U << Reg1.GetCode()) & FPRFillMask)) {
ldr(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
}
else if (((1U << Reg2.GetCode()) & FPRFillMask)) {
ldr(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
}
}
}
}
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
ldp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
}
void Arm64Emitter::PushDynamicRegsAndLR() {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
const auto GPRSize = (RA64.size() + 1) * Core::CPUState::GPR_REG_SIZE;
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
: Core::CPUState::XMM_SSE_REG_SIZE;
const auto FPRSize = RAFPR.size() * FPRRegSize;
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
sub(sp, sp, SPOffset);
int i = 0;
if (CanUseSVE) {
for (const auto& RA : RAFPR) {
mov(TMP4, i * 8);
st1b(RA.Z().VnB(), PRED_TMP_32B, SVEMemOperand(sp, TMP4));
i += 4;
}
} else {
for (const auto& RA : RAFPR) {
str(RA.Q(), MemOperand(sp, i * 8));
i += 2;
}
for (auto RA : RAFPR)
{
str(RA.Q(), MemOperand(sp, i * 8));
i+=2;
}
#if 0 // All GPRs should be caller saved
for (const auto& RA : RA64) {
for (auto RA : RA64)
{
str(RA, MemOperand(sp, i * 8));
i++;
}
@@ -341,29 +175,18 @@ void Arm64Emitter::PushDynamicRegsAndLR() {
}
void Arm64Emitter::PopDynamicRegsAndLR() {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
const auto GPRSize = (RA64.size() + 1) * Core::CPUState::GPR_REG_SIZE;
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
: Core::CPUState::XMM_SSE_REG_SIZE;
const auto FPRSize = RAFPR.size() * FPRRegSize;
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
int i = 0;
if (CanUseSVE) {
for (const auto& RA : RAFPR) {
mov(TMP4, i * 8);
ld1b(RA.Z().VnB(), PRED_TMP_32B.Zeroing(), SVEMemOperand(sp, TMP4));
i += 4;
}
} else {
for (const auto& RA : RAFPR) {
ldr(RA.Q(), MemOperand(sp, i * 8));
i += 2;
}
for (auto RA : RAFPR)
{
ldr(RA.Q(), MemOperand(sp, i * 8));
i+=2;
}
#if 0 // All GPRs should be caller saved
for (const auto& RA : RA64) {
for (auto RA : RA64)
{
ldr(RA, MemOperand(sp, i * 8));
i++;
}
@@ -374,10 +197,23 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
add(sp, sp, SPOffset);
}
void Arm64Emitter::ResetStack() {
if (SpillSlots == 0)
return;
if (IsImmAddSub(SpillSlots * 16)) {
add(sp, sp, SpillSlots * 16);
} else {
// Too big to fit in a 12bit immediate
LoadConstant(x0, SpillSlots * 16);
add(sp, sp, x0);
}
}
void Arm64Emitter::Align16B() {
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
uint64_t CurrentOffset = GetBuffer()->GetOffsetAddress<uint64_t>(GetCursorOffset());
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
nop();
nop();
}
}
@@ -1,24 +1,7 @@
#pragma once
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/ObjectCache/Relocations.h"
#include <aarch64/assembler-aarch64.h>
#include <aarch64/constants-aarch64.h>
#include <aarch64/cpu-aarch64.h>
#include <aarch64/operands-aarch64.h>
#include <platform-vixl.h>
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#include <aarch64/simulator-constants-aarch64.h>
#endif
#include <FEXCore/Config/Config.h>
#include <array>
#include <cstddef>
#include <cstdint>
#include <utility>
#include "aarch64/assembler-aarch64.h"
#include "aarch64/cpu-aarch64.h"
namespace FEXCore::CPU {
using namespace vixl;
@@ -62,49 +45,19 @@ const std::array<aarch64::VRegister, 12> RAFPR = {
v8, v9, v10, v11, v12, v13, v14, v15
};
// Contains the address to the currently available CPU state
#define STATE x28
// GPR temporaries. Only x3 can be used across spill boundaries
// so if these ever need to change, be very careful about that.
#define TMP1 x0
#define TMP2 x1
#define TMP3 x2
#define TMP4 x3
// Vector temporaries
#define VTMP1 v1
#define VTMP2 v2
#define VTMP3 v3
// Predicate register temporaries (used when AVX support is enabled)
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
#define PRED_TMP_16B p6
#define PRED_TMP_32B p7
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
class Arm64Emitter : public vixl::aarch64::Assembler {
protected:
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
~Arm64Emitter();
Arm64Emitter(size_t size);
FEXCore::Context::Context *EmitterCTX;
vixl::aarch64::CPU CPU;
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
bool SupportsAtomics{};
bool SupportsRCPC{};
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
// TMP4 is left alone.
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
static constexpr uint32_t CALLER_GPR_MASK = 0b0011'1111'1111'1111'1111;
// This isn't technically true because the lower 64-bits of v8..v15 are callee saved
// We can't guarantee only the lower 64bits are used so flush everything
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant);
void SpillStaticRegs();
void FillStaticRegs();
void PushDynamicRegsAndLR();
void PopDynamicRegsAndLR();
@@ -112,74 +65,10 @@ protected:
void PushCalleeSavedRegisters();
void PopCalleeSavedRegisters();
void ResetStack();
void Align16B();
#ifdef VIXL_SIMULATOR
// Generates a vixl simulator runtime call.
//
// This matches behaviour of vixl's macro assembler, but we need to reimplement it since we aren't using the macro assembler.
// This isn't too complex with how vixl emits this.
//
// Emit:
// 1) hlt(kRuntimeCallOpcode)
// 2) Simulator wrapper handler
// 3) Function to call
// 4) Style of the function call (Call versus tail-call)
template<typename R, typename... P>
void GenerateRuntimeCall(R (*Function)(P...)) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
uintptr_t FunctionAddress = reinterpret_cast<uintptr_t>(Function);
hlt(kRuntimeCallOpcode);
// Simulator wrapper address pointer.
dc(SimulatorWrapperAddress);
// Runtime function address to call
dc(FunctionAddress);
// Call type
dc32(kCallRuntime);
}
template<typename R, typename... P>
void GenerateIndirectRuntimeCall(vixl::aarch64::Register Reg) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
hlt(kIndirectRuntimeCallOpcode);
// Simulator wrapper address pointer.
dc(SimulatorWrapperAddress);
// Register that contains the function to call
dc(Reg.GetCode());
// Call type
dc32(kCallRuntime);
}
template<>
void GenerateIndirectRuntimeCall<float, __uint128_t>(vixl::aarch64::Register Reg) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
hlt(kIndirectRuntimeCallOpcode);
// Simulator wrapper address pointer.
dc(SimulatorWrapperAddress);
// Register that contains the function to call
dc(Reg.GetCode());
// Call type
dc32(kCallRuntime);
}
#endif
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
uint32_t SpillSlots{};
};
}
@@ -1,7 +1,6 @@
#include "Interface/Core/ArchHelpers/Arm64.h"
#include <FEXCore/Utils/LogManager.h>
#include <stdint.h>
namespace FEXCore::ArchHelpers::Arm64 {
@@ -12,16 +11,16 @@ namespace FEXCore::ArchHelpers::Arm64 {
// Obvously such a configuration can't do the actual arm64-specific stuff
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
ERROR_AND_DIE_FMT("HandleCASPAL Not Implemented");
ERROR_AND_DIE("HandleCASPAL Not Implemented");
}
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
ERROR_AND_DIE_FMT("HandleCASAL Not Implemented");
ERROR_AND_DIE("HandleCASAL Not Implemented");
}
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
ERROR_AND_DIE("HandleAtomicMemOp Not Implemented");
}
#endif
}
}
+22 -77
View File
@@ -13,27 +13,16 @@
namespace FEXCore::ArchHelpers::Context {
enum ContextFlags : uint32_t {
CONTEXT_FLAG_INJIT = (1U << 0),
CONTEXT_FLAG_32BIT = (1U << 1),
};
struct X86ContextBackup {
// Host State
// RIP and RSP is stored in GPRs here
uint64_t GPRs[23];
FEXCore::x86_64::_libc_fpstate FPRState;
uint64_t sa_mask;
bool FaultToTopAndGeneratedException;
// Guest state
int Signal;
uint32_t Flags;
uint64_t OriginalRIP;
uint64_t FPStateLocation;
uint64_t UContextLocation;
uint64_t SigInfoLocation;
FEXCore::Core::CPUState GuestState;
static constexpr int RedZoneSize = 128;
};
@@ -46,27 +35,15 @@ struct ArmContextBackup {
uint32_t FPSR;
uint32_t FPCR;
__uint128_t FPRs[32];
uint64_t sa_mask;
bool FaultToTopAndGeneratedException;
// Guest state
int Signal;
uint32_t Flags;
uint64_t OriginalRIP;
uint64_t FPStateLocation;
uint64_t UContextLocation;
uint64_t SigInfoLocation;
FEXCore::Core::CPUState GuestState;
// Arm64 doesn't have a red zone
static constexpr int RedZoneSize = 0;
};
static inline ucontext_t* GetUContext(void* ucontext) {
ucontext_t* _context = (ucontext_t*)ucontext;
return _context;
}
static inline mcontext_t* GetMContext(void* ucontext) {
ucontext_t* _context = (ucontext_t*)ucontext;
return &_context->uc_mcontext;
@@ -75,20 +52,6 @@ static inline mcontext_t* GetMContext(void* ucontext) {
#ifdef _M_ARM_64
constexpr uint32_t FPR_MAGIC = 0x46508001U;
struct HostCTXHeader {
uint32_t Magic;
uint32_t Size;
};
struct HostFPRState {
HostCTXHeader Head;
uint32_t FPSR;
uint32_t FPCR;
__uint128_t FPRs[32];
};
static inline uint64_t GetSp(void* ucontext) {
return GetMContext(ucontext)->sp;
}
@@ -121,19 +84,24 @@ static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
GetMContext(ucontext)->regs[id] = val;
}
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
auto MContext = GetMContext(ucontext);
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&MContext->__reserved[0]);
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
constexpr uint32_t FPR_MAGIC = 0x46508001U;
return HostState->FPRs[id];
}
struct HostCTXHeader {
uint32_t Magic;
uint32_t Size;
};
struct HostFPRState {
HostCTXHeader Head;
uint32_t FPSR;
uint32_t FPCR;
__uint128_t FPRs[32];
};
using ContextBackup = ArmContextBackup;
template <typename T>
static inline void BackupContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, ArmContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
memcpy(&Backup->GPRs[0], &_mcontext->regs[0], 31 * sizeof(uint64_t));
@@ -143,27 +111,22 @@ static inline void BackupContext(void* ucontext, T *Backup) {
// Host FPR state starts at _mcontext->reserved[0];
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
LogMan::Throw::A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
Backup->FPSR = HostState->FPSR;
Backup->FPCR = HostState->FPCR;
memcpy(&Backup->FPRs[0], &HostState->FPRs[0], 32 * sizeof(__uint128_t));
// Save the signal mask so we can restore it
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
template <typename T>
static inline void RestoreContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, ArmContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
LogMan::Throw::A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
memcpy(&HostState->FPRs[0], &Backup->FPRs[0], 32 * sizeof(__uint128_t));
HostState->FPCR = Backup->FPCR;
HostState->FPSR = Backup->FPSR;
@@ -173,12 +136,8 @@ static inline void RestoreContext(void* ucontext, T *Backup) {
ArchHelpers::Context::SetPc(ucontext, Backup->PrevPC);
ArchHelpers::Context::SetSp(ucontext, Backup->PrevSP);
memcpy(&_mcontext->regs[0], &Backup->GPRs[0], 31 * sizeof(uint64_t));
// Restore the signal mask now
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
@@ -211,22 +170,17 @@ static inline void SetState(void* ucontext, uint64_t val) {
}
static inline uint64_t GetArmReg(void* ucontext, uint32_t id) {
ERROR_AND_DIE_FMT("Not impelented for x86 host");
ERROR_AND_DIE("Not impelented for x86 host");
}
static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
ERROR_AND_DIE_FMT("Not impelented for x86 host");
}
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
ERROR_AND_DIE_FMT("Not implemented for x86 host");
ERROR_AND_DIE("Not impelented for x86 host");
}
using ContextBackup = X86ContextBackup;
template <typename T>
static inline void BackupContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, X86ContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
// Copy the GPRs
@@ -234,34 +188,25 @@ static inline void BackupContext(void* ucontext, T *Backup) {
// Copy the FPRState
memcpy(&Backup->FPRState, _mcontext->fpregs, sizeof(X86ContextBackup::FPRState));
// XXX: Save 256bit and 512bit AVX register state
// Save the signal mask so we can restore it
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
template <typename T>
static inline void RestoreContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, X86ContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
// Copy the GPRs
memcpy(&_mcontext->gregs[0], &Backup->GPRs[0], sizeof(X86ContextBackup::GPRs));
// Copy the FPRState
memcpy(_mcontext->fpregs, &Backup->FPRState, sizeof(X86ContextBackup::FPRState));
// Restore the signal mask now
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
#endif
} // namespace FEXCore::ArchHelpers::Context
} // namespace FEXCore::ArchHelpers::Context
@@ -2,7 +2,6 @@
#include <FEXCore/Utils/LogManager.h>
#include <cstring>
#include <fstream>
#include <utility>
namespace FEXCore {
void BlockSamplingData::DumpBlockData() {
@@ -27,7 +26,7 @@ namespace FEXCore {
<< std::endl;
}
Output.close();
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
LogMan::Msg::D("Dumped %d blocks of sampling data", SamplingMap.size());
}
BlockSamplingData::BlockData *BlockSamplingData::GetBlockData(uint64_t RIP) {
+1 -2
View File
@@ -1,7 +1,6 @@
#pragma once
#include <cstdint>
#include <unordered_map>
#include <stdint.h>
namespace FEXCore {
class BlockSamplingData {
-87
View File
@@ -1,87 +0,0 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include <FEXCore/Core/CPUBackend.h>
namespace FEXCore {
namespace CPU {
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {}
CPUBackend::~CPUBackend() {
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
}
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
if (CodeBuffers.empty()) {
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
} else {
if (CodeBuffers.size() > 1) {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
for (size_t i = 1; i < CodeBuffers.size(); i++) {
FreeCodeBuffer(CodeBuffers[i]);
}
CodeBuffers.resize(1);
}
// Set the current code buffer to the initial
CurrentCodeBuffer = &CodeBuffers[0];
if (CurrentCodeBuffer->Size != MaxCodeSize) {
FreeCodeBuffer(*CurrentCodeBuffer);
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer->Size *= 1.5;
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
}
}
} else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
}
return CurrentCodeBuffer;
}
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t *>(
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
if (ThreadState->CTX->Config.GlobalJITNaming()) {
ThreadState->CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
for (auto &Buffer: CodeBuffers) {
auto start = (uintptr_t)Buffer.Ptr;
auto end = start + Buffer.Size;
if (Address >= start && Address < end) {
return true;
}
}
return false;
}
}
}
File diff suppressed because it is too large. Load diff
+26 -66
View File
@@ -1,12 +1,9 @@
#pragma once
#include <cstdint>
#include <functional>
#include <unordered_map>
#include <utility>
#include <vector>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
namespace FEXCore {
namespace Context {
@@ -24,83 +21,46 @@ private:
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
public:
// X86 cacheline size effectively has to be hardcoded to 64
// if we report anything differently then applications are likely to break
constexpr static uint64_t CACHELINE_SIZE = 64;
void Init(FEXCore::Context::Context *ctx);
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) {
const auto Handler = FunctionHandlers.find(Function);
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, [[maybe_unused]] uint32_t Leaf) {
auto Handler = FunctionHandlers.find(Function);
if (Handler == FunctionHandlers.end()) {
return Function_Reserved(Leaf);
#ifndef NDEBUG
LogMan::Msg::E("Unhandled CPU ID function, 0x%x", Function);
#endif
return Function_Reserved();
}
return (this->*Handler->second)(Leaf);
return Handler->second();
}
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
if (Function == 0x8000'0002U)
return Function_8000_0002h(Leaf, CPU % PerCPUData.size());
else if (Function == 0x8000'0003U)
return Function_8000_0003h(Leaf, CPU % PerCPUData.size());
else
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
}
private:
FEXCore::Context::Context *CTX;
bool Hybrid{};
FEX_CONFIG_OPT(Cores, THREADS);
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
using FunctionHandler = std::function<FEXCore::CPUID::FunctionResults()>;
void RegisterFunction(uint32_t Function, FunctionHandler Handler) {
FunctionHandlers.insert_or_assign(Function, Handler);
FunctionHandlers[Function] = Handler;
}
std::unordered_map<uint32_t, FunctionHandler> FunctionHandlers;
struct CPUData {
const char *ProductName{};
#ifdef _M_ARM_64
uint32_t MIDR{};
#endif
bool IsBig{};
};
std::vector<CPUData> PerCPUData{};
// Functions
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_01h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_02h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_04h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_06h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_07h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_4000_0001h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf, uint32_t CPU);
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf, uint32_t CPU);
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf, uint32_t CPU);
FEXCore::CPUID::FunctionResults Function_8000_0005h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0008h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0009h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
void SetupHostHybridFlag();
FEXCore::CPUID::FunctionResults Function_0h();
FEXCore::CPUID::FunctionResults Function_01h();
FEXCore::CPUID::FunctionResults Function_02h();
FEXCore::CPUID::FunctionResults Function_06h();
FEXCore::CPUID::FunctionResults Function_07h();
FEXCore::CPUID::FunctionResults Function_15h();
FEXCore::CPUID::FunctionResults Function_8000_0000h();
FEXCore::CPUID::FunctionResults Function_8000_0001h();
FEXCore::CPUID::FunctionResults Function_8000_0002h();
FEXCore::CPUID::FunctionResults Function_8000_0003h();
FEXCore::CPUID::FunctionResults Function_8000_0004h();
FEXCore::CPUID::FunctionResults Function_8000_0005h();
FEXCore::CPUID::FunctionResults Function_8000_0006h();
FEXCore::CPUID::FunctionResults Function_8000_0007h();
FEXCore::CPUID::FunctionResults Function_Reserved();
};
}
@@ -0,0 +1,168 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/CompileService.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/OpcodeDispatcher.h"
namespace FEXCore {
static void* ThreadHandler(void *Arg) {
FEXCore::CompileService *This = reinterpret_cast<FEXCore::CompileService*>(Arg);
This->ExecutionThread();
return nullptr;
}
CompileService::CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
: CTX {ctx}
, ParentThread {Thread} {
CompileThreadData = std::make_unique<FEXCore::Core::InternalThreadState>();
CompileThreadData->IsCompileService = true;
// We need a compiler for this work thread
CTX->InitializeCompiler(CompileThreadData.get(), true);
CompileThreadData->CPUBackend->CopyNecessaryDataForCompileThread(ParentThread->CPUBackend.get());
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
}
void CompileService::Initialize() {
// Share CompileService which = this
CompileThreadData->CompileService = ParentThread->CompileService;
}
void CompileService::Shutdown() {
ShuttingDown = true;
// Kick the working thread
StartWork.NotifyAll();
WorkerThread->join(nullptr);
}
void CompileService::ClearCache(FEXCore::Core::InternalThreadState *Thread) {
// On cache clear we need to spin down the execution thread to ensure it isn't trying to give us more work items
if (CompileMutex.try_lock()) {
// We can only clear these things if we pulled the compile mutex
// Grab the work queue and clear it
// We don't need to grab the queue mutex since this thread will no longer receive any work events
// Threads are bounded 1:1
while (WorkQueue.size()) {
WorkItem *Item = WorkQueue.front();
WorkQueue.pop();
delete Item;
}
// Go through the garbage collection array and clear it
// It's safe to clear things that aren't marked safe since we are clearing cache
if (GCArray.size()) {
// Clean up our GC array
for (auto it = GCArray.begin(); it != GCArray.end();) {
delete *it;
it = GCArray.erase(it);
}
}
LogMan::Throw::A(CompileThreadData->LocalIRCache.size() == 0, "Compile service must never have LocalIRCache");
CompileMutex.unlock();
}
// Clear the inverse cache of what is calling us from the Context ClearCache routine
auto SelectedThread = Thread->IsCompileService ? ParentThread : Thread;
SelectedThread->LookupCache->ClearCache();
SelectedThread->CPUBackend->ClearCache();
}
CompileService::WorkItem *CompileService::CompileCode(uint64_t RIP) {
// Tell the worker thread to compile code for us
WorkItem *Item = new WorkItem{};
Item->RIP = RIP;
{
// Fill the threads work queue
std::scoped_lock<std::mutex> lk(QueueMutex);
WorkQueue.emplace(Item);
}
// Notify the thread that it has more work
StartWork.NotifyAll();
return Item;
}
void CompileService::ExecutionThread() {
// Ignore signals coming from the guest
CTX->SignalDelegation->MaskThreadSignals();
// Set our thread name so we can see its relation
char ThreadName[16]{};
snprintf(ThreadName, 16, "%ld-CS", ParentThread->ThreadManager.TID.load());
pthread_setname_np(pthread_self(), ThreadName);
while (true) {
// Wait for work
StartWork.Wait();
if (ShuttingDown.load()) {
break;
}
std::scoped_lock<std::mutex> lk(CompileMutex);
size_t WorkItems{};
do {
// Grab a work item
WorkItem *Item{};
{
std::scoped_lock<std::mutex> lk(QueueMutex);
WorkItems = WorkQueue.size();
if (WorkItems) {
Item = WorkQueue.front();
WorkQueue.pop();
}
}
// If we had a work item then work on it
if (Item) {
// Make sure it's not in lookup cache by accident
LogMan::Throw::A(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
// Code isn't in cache, compile now
// Set our thread state's RIP
CompileThreadData->CurrentFrame->State.rip = Item->RIP;
auto [CodePtr, IRList, DebugData, RAData, Generated, StartAddr, Length] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
LogMan::Throw::A(Generated == true, "Compile Service doesn't have IR Cache");
if (!CodePtr) {
// XXX: We currently have the expectation that compile service code will be significantly smaller than regular thread's code
ERROR_AND_DIE("Couldn't compile code for thread at RIP: 0x%lx", Item->RIP);
}
Item->CodePtr = CodePtr;
Item->IRList = IRList;
Item->DebugData = DebugData;
Item->RAData = RAData;
Item->StartAddr = StartAddr;
Item->Length = Length;
GCArray.emplace_back(Item);
Item->ServiceWorkDone.NotifyAll();
}
} while (WorkItems != 0);
if (GCArray.size()) {
// Clean up our GC array
for (auto it = GCArray.begin(); it != GCArray.end();) {
if ((*it)->SafeToClear) {
delete *it;
it = GCArray.erase(it);
}
else {
++it;
}
}
}
}
}
}
+66
View File
@@ -0,0 +1,66 @@
#pragma once
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/Threads.h>
#include <memory>
#include <thread>
#include <unordered_map>
#include <queue>
namespace FEXCore {
namespace Context {
struct Context;
}
namespace Core {
struct InternalThreadState;
}
namespace IR {
class RegisterAllocationData;
};
class CompileService final {
public:
CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
void Initialize();
void Shutdown();
struct WorkItem {
// Incoming
uint64_t RIP{};
// Outgoing
void *CodePtr{};
FEXCore::IR::IRListView *IRList{};
FEXCore::IR::RegisterAllocationData *RAData{};
FEXCore::Core::DebugData *DebugData{};
uint64_t StartAddr;
uint64_t Length;
// Communication
Event ServiceWorkDone{};
std::atomic_bool SafeToClear{};
};
WorkItem *CompileCode(uint64_t RIP);
void ClearCache(FEXCore::Core::InternalThreadState *Thread);
// Public for threading
void ExecutionThread();
private:
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *ParentThread;
std::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
std::unique_ptr<FEXCore::Core::InternalThreadState> CompileThreadData;
std::mutex QueueMutex{};
std::mutex CompileMutex{};
std::queue<WorkItem*> WorkQueue{};
std::vector<WorkItem*> GCArray{};
Event StartWork{};
std::atomic_bool ShuttingDown{false};
};
}
File diff suppressed because it is too large. Load diff
@@ -1,60 +1,31 @@
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/X86HelperGen.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <array>
#include <bit>
#include <cmath>
#include <cstddef>
#include <cstdint>
#include <memory>
#include <aarch64/assembler-aarch64.h>
#include <aarch64/constants-aarch64.h>
#include <aarch64/cpu-aarch64.h>
#include <aarch64/operands-aarch64.h>
#include <code-buffer-vixl.h>
#include <platform-vixl.h>
#include <sys/syscall.h>
#include <unistd.h>
#define STATE_PTR(STATE_TYPE, FIELD) \
MemOperand(STATE, offsetof(FEXCore::Core::STATE_TYPE, FIELD))
#include "aarch64/assembler-aarch64.h"
#include "aarch64/cpu-aarch64.h"
#include "aarch64/disasm-aarch64.h"
namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 8192;
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
#ifdef VIXL_SIMULATOR
, Simulator {&Decoder}
#endif
{
#ifdef VIXL_SIMULATOR
// Hardcode a 256-bit vector width if we are running in the simulator.
Simulator.SetVectorLengthInBits(256);
#endif
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
#define STATE x28
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
: Dispatcher(ctx, Thread), Arm64Emitter(MAX_DISPATCHER_CODE_SIZE) {
SRAEnabled = config.StaticRegisterAssignment;
SetAllowAssembler(true);
DispatchPtr = GetCursorAddress<AsmDispatch>();
auto Buffer = GetBuffer();
DispatchPtr = Buffer->GetOffsetAddress<CPUBackend::AsmDispatch>(GetCursorOffset());
// while (true) {
// Ptr = FindBlock(RIP)
@@ -64,9 +35,15 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
// Ptr();
// }
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
Literal l_VirtualMemory {VirtualMemorySize};
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
Literal l_L1Ptr {Thread->LookupCache->GetL1Pointer()};
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
Literal l_CompileBlock {GetCompileBlockPtr()};
Literal l_ExitFunctionLink {config.ExitFunctionLink};
Literal l_ExitFunctionLinkThis {config.ExitFunctionLinkThis};
// Push all the register we need to save
PushCalleeSavedRegisters();
@@ -79,18 +56,17 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
// regardless of where we were in the stack
add(x0, sp, 0);
str(x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
AbsoluteLoopTopAddressFillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (config.StaticRegisterAllocation) {
if (SRAEnabled) {
FillStaticRegs();
}
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
aarch64::Label FullLookup{};
aarch64::Label CallBlock{};
aarch64::Label LoopTop{};
aarch64::Label ExitSpillSRA{};
aarch64::Label ThreadPauseHandler{};
@@ -100,34 +76,34 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
// Load in our RIP
// Don't modify x2 since it contains our RIP once the block doesn't exist
ldr(x2, STATE_PTR(CpuStateFrame, State.rip));
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
auto RipReg = x2;
// L1 Cache
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
if (!config.ExecuteBlocksWithCall) {
// L1 Cache
ldr(x0, &l_L1Ptr);
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x3, Shift::LSL, 4));
ldp(x3, x0, MemOperand(x0));
cmp(x0, RipReg);
b(&FullLookup, Condition::ne);
br(x3);
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x3, Shift::LSL, 4));
ldp(x1, x0, MemOperand(x0));
cmp(x0, RipReg);
b(&FullLookup, Condition::ne);
br(x1);
}
// L1C check failed, do a full lookup
bind(&FullLookup);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
ldr(x0, &l_PagePtr);
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(x3, RipReg, VirtualMemorySize - 1);
if (__builtin_popcountl(VirtualMemorySize) == 1) {
and_(x3, RipReg, Thread->LookupCache->GetVirtualMemorySize() - 1);
}
else {
LoadConstant(x3, VirtualMemorySize);
ldr(x3, &l_VirtualMemory);
and_(x3, RipReg, x3);
}
@@ -160,25 +136,51 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x1, Shift::LSL, 4));
stp(x3, x2, MemOperand(x0));
// Jump to the block
br(x3);
if (!config.ExecuteBlocksWithCall) {
// update L1 cache
ldr(x0, &l_L1Ptr);
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x1, Shift::LSL, 4));
stp(x3, x2, MemOperand(x0));
br(x3);
} else {
mov(x0, STATE);
blr(x3);
}
}
if (config.ExecuteBlocksWithCall) {
// Interpreter continues execution here
if (CTX->GetGdbServerStatus()) {
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
ldr(x0, &l_CTX);
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
// If the value == 0 then branch to the top
cbz(x0, &LoopTop);
// Else we need to pause now
b(&ThreadPauseHandler);
}
else {
// Unconditionally loop to the top
// We will only stop on error when compiling a block or signal
b(&LoopTop);
}
}
}
{
bind(&ExitSpillSRA);
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
ThreadStopHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled)
SpillStaticRegs();
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
ThreadStopHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
PopCalleeSavedRegisters();
@@ -187,65 +189,19 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
ret();
}
#ifdef VIXL_SIMULATOR
// VIXL simulator can't run syscalls.
constexpr bool SignalSafeCompile = false;
#else
constexpr bool SignalSafeCompile = true;
#endif
{
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
ExitFunctionLinkerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled)
SpillStaticRegs();
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// Args:
// X0: SETMASK
// X1: Pointer to mask value (uint64_t)
// X2: Pointer to old mask value (uint64_t)
// X3: Size of mask, sizeof(uint64_t)
// X8: Syscall
ldr(x0, &l_ExitFunctionLinkThis);
mov(x1, STATE);
mov(x2, lr);
LoadConstant(x0, ~0ULL);
stp(x0, x0, MemOperand(sp, -16, PreIndex));
LoadConstant(x0, SIG_SETMASK);
add(x1, sp, 0);
add(x2, sp, 0);
LoadConstant(x3, 8);
LoadConstant(x8, SYS_rt_sigprocmask);
svc(0);
}
ldr(x3, &l_ExitFunctionLink);
blr(x3);
mov(x0, STATE);
mov(x1, lr);
ldr(x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uintptr_t, void *, void *>(x2);
#else
blr(x2);
#endif
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
mov(x4, x0);
LoadConstant(x0, SIG_SETMASK);
add(x1, sp, 0);
LoadConstant(x2, 0);
LoadConstant(x3, 8);
LoadConstant(x8, SYS_rt_sigprocmask);
svc(0);
// Bring stack back
add(sp, sp, 16);
mov(x0, x4);
}
if (config.StaticRegisterAllocation)
if (SRAEnabled)
FillStaticRegs();
br(x0);
}
@@ -254,64 +210,24 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
{
bind(&NoBlock);
if (config.StaticRegisterAllocation)
SpillStaticRegs();
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// Args:
// X0: SETMASK
// X1: Pointer to mask value (uint64_t)
// X2: Pointer to old mask value (uint64_t)
// X3: Size of mask, sizeof(uint64_t)
// X8: Syscall
LoadConstant(x0, ~0ULL);
stp(x0, x2, MemOperand(sp, -16, PreIndex));
LoadConstant(x0, SIG_SETMASK);
add(x1, sp, 0);
add(x2, sp, 0);
LoadConstant(x3, 8);
LoadConstant(x8, SYS_rt_sigprocmask);
svc(0);
// Reload x2 to bring back RIP
ldr(x2, MemOperand(sp, 8, Offset));
}
ldr(x0, &l_CTX);
mov(x1, STATE);
ldr(x3, &l_CompileBlock);
if (SRAEnabled)
SpillStaticRegs();
// X2 contains our guest RIP
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<void, void *, uint64_t, void *>(x3);
#else
blr(x3); // { CTX, Frame, RIP}
#endif
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
LoadConstant(x0, SIG_SETMASK);
add(x1, sp, 0);
LoadConstant(x2, 0);
LoadConstant(x3, 8);
LoadConstant(x8, SYS_rt_sigprocmask);
svc(0);
// Bring stack back
add(sp, sp, 16);
}
if (config.StaticRegisterAllocation)
if (SRAEnabled)
FillStaticRegs();
b(&LoopTop);
}
{
SignalHandlerReturnAddress = GetCursorAddress<uint64_t>();
SignalHandlerReturnAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
// Now to get back to our old location we need to do a fault dance
// We can't use SIGTRAP here since gdb catches it and never gives it to the application!
@@ -319,50 +235,12 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
}
{
// Guest SIGILL handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
SpillStaticRegs();
hlt(0);
}
{
// Guest SIGTRAP handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
SpillStaticRegs();
brk(0);
}
{
// Guest Overflow handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
SpillStaticRegs();
// hlt/udf = SIGILL
// brk = SIGTRAP
// ??? = SIGSEGV
// Force a SIGSEGV by loading zero
LoadConstant(x1, 0);
ldr(x1, MemOperand(x1));
}
{
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
ThreadPauseHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled)
SpillStaticRegs();
bind(&ThreadPauseHandler);
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
ThreadPauseHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
// We are pausing, this means the frontend should be waiting for this thread to idle
// We will have faulted and jumped to this location at this point
@@ -370,13 +248,9 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
ldr(x0, &l_CTX);
mov(x1, STATE);
ldr(x2, &l_Sleep);
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<void, void *, void *>(x2);
#else
blr(x2);
#endif
PauseReturnInstruction = GetCursorAddress<uint64_t>();
PauseReturnInstruction = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
// Fault to start running again
hlt(0);
}
@@ -397,7 +271,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
// On return to the thunk, the thunk can get whatever its return value is from the thread context depending on ABI handling on its end
// When the thunk itself returns, it'll do its regular return logic there
// void ReentrantCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
CallbackPtr = GetCursorAddress<JITCallback>();
CallbackPtr = Buffer->GetOffsetAddress<CPUBackend::JITCallback>(GetCursorOffset());
// We expect the thunk to have previously pushed the registers it was using
PushCalleeSavedRegisters();
@@ -406,269 +280,84 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
mov(STATE, x0);
// Make sure to adjust the refcounter so we don't clear the cache now
ldr(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
LoadConstant(x0, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
ldr(w2, MemOperand(x0));
add(w2, w2, 1);
str(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
str(w2, MemOperand(x0));
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
LoadConstant(x0, CTX->X86CodeGen.CallbackReturn);
ldr(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
sub(x2, x2, 16);
str(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
str(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
// Store the trampoline to the guest stack
// Guest stack is now correctly misaligned after a regular call instruction
str(x0, MemOperand(x2));
// Store RIP to the context state
str(x1, STATE_PTR(CpuStateFrame, State.rip));
str(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
// load static regs
if (config.StaticRegisterAllocation)
if (SRAEnabled)
FillStaticRegs();
// Now go back to the regular dispatcher loop
b(&LoopTop);
}
{
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR();
SpillStaticRegs();
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
#else
blr(x3);
#endif
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR();
SpillStaticRegs();
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
#else
blr(x3);
#endif
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR();
SpillStaticRegs();
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
#else
blr(x3);
#endif
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR();
SpillStaticRegs();
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
#else
blr(x3);
#endif
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
place(&l_VirtualMemory);
place(&l_PagePtr);
place(&l_L1Ptr);
place(&l_CTX);
place(&l_Sleep);
place(&l_CompileBlock);
place(&l_ExitFunctionLink);
place(&l_ExitFunctionLinkThis);
FinalizeCode();
Start = reinterpret_cast<uint64_t>(DispatchPtr);
End = GetCursorAddress<uint64_t>();
End = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
GetBuffer()->SetExecutable();
if (CTX->Config.BlockJITNaming()) {
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
}
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
}
#if ENABLE_JITSYMBOLS
std::string Name = "Dispatch_" + std::to_string(::gettid());
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
#endif
}
#ifdef VIXL_SIMULATOR
void Arm64Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.RunFrom(reinterpret_cast<Instruction const*>(DispatchPtr));
void Arm64Dispatcher::SpillSRA(void *ucontext) {
for(int i = 0; i < SRA64.size(); i++) {
ThreadState->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
}
// TODO: Also recover FPRs, not sure where the neon context is
// This is usually not needed
/*
for(int i = 0; i < SRAFPR.size(); i++) {
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
}
*/
}
void Arm64Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.WriteXRegister(1, RIP);
Simulator.RunFrom(reinterpret_cast<Instruction const*>(CallbackPtr));
#ifdef _M_ARM_64
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
DispatcherConfig config;
config.ExecuteBlocksWithCall = true;
Dispatcher = new Arm64Dispatcher(ctx, Thread, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
// TODO: It feels wrong to initialize this way
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
}
#endif
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline, destination buffer is set before use
static thread_local vixl::aarch64::Assembler emit((uint8_t*)&emit, 1);
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxGDBPauseCheckSize);
vixl::CodeBufferCheckScope scope(&emit, MaxGDBPauseCheckSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
aarch64::Label RunBlock;
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(FEXCore::Context::Context::Config.RunningMode) == 4, "This is expected to be size of 4");
emit.ldr(x0, STATE_PTR(CpuStateFrame, Thread)); // Get thread
emit.ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
emit.ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
// If the value == 0 then we don't need to stop
emit.cbz(w0, &RunBlock);
{
Literal l_GuestRIP {GuestRIP};
// Make sure RIP is syncronized to the context
emit.ldr(x0, &l_GuestRIP);
emit.str(x0, STATE_PTR(CpuStateFrame, State.rip));
// Stop the thread
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
emit.br(x0);
emit.place(&l_GuestRIP);
}
emit.bind(&RunBlock);
emit.FinalizeCode();
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
return UsedBytes;
}
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
vixl::CodeBufferCheckScope scope(&emit, MaxInterpreterTrampolineSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
aarch64::Label InlineIRData;
emit.mov(x0, STATE);
emit.adr(x1, &InlineIRData);
emit.ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
emit.blr(x3);
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
emit.br(x0);
emit.bind(&InlineIRData);
emit.FinalizeCode();
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
return UsedBytes;
}
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
for (size_t i = 0; i < SRA64.size(); i++) {
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
// Skip this one, it's already spilled
continue;
}
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
}
if (EmitterCTX->HostFeatures.SupportsAVX) {
for (size_t i = 0; i < SRAFPR.size(); i++) {
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &FPR, sizeof(__uint128_t));
}
} else {
for (size_t i = 0; i < SRAFPR.size(); i++) {
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &FPR, sizeof(__uint128_t));
}
}
}
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto &Common = Thread->CurrentFrame->Pointers.Common;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
Common.SignalReturnHandler = SignalHandlerReturnAddress;
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
AArch64.LUDIVHandler = LUDIVHandlerAddress;
AArch64.LDIVHandler = LDIVHandlerAddress;
AArch64.LUREMHandler = LUREMHandlerAddress;
AArch64.LREMHandler = LREMHandlerAddress;
}
}
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
return std::make_unique<Arm64Dispatcher>(CTX, Config);
}
}
@@ -3,46 +3,16 @@
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#endif
namespace FEXCore::Context {
struct Context;
}
namespace FEXCore::Core {
struct InternalThreadState;
}
#include "aarch64/assembler-aarch64.h"
namespace FEXCore::CPU {
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
public:
Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
#ifdef VIXL_SIMULATOR
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) override;
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) override;
#endif
Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
protected:
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
private:
// Long division helpers
uint64_t LUDIVHandlerAddress{};
uint64_t LDIVHandlerAddress{};
uint64_t LUREMHandlerAddress{};
uint64_t LREMHandlerAddress{};
#ifdef VIXL_SIMULATOR
vixl::aarch64::Decoder Decoder;
vixl::aarch64::Simulator Simulator;
#endif
void SpillSRA(void *ucontext) override;
};
}
}
+111 -610
View File
@@ -1,24 +1,8 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86HelperGen.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/UContext.h>
#include "Common/MathUtils.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <atomic>
#include <condition_variable>
#include <csignal>
#include <cstring>
#include <signal.h>
namespace FEXCore::CPU {
@@ -28,19 +12,15 @@ void Dispatcher::SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuS
--ctx->IdleWaitRefCount;
ctx->IdleWaitCV.notify_all();
Thread->RunningEvents.ThreadSleeping = true;
// Go to sleep
Thread->StartRunning.Wait();
Thread->RunningEvents.Running = true;
++ctx->IdleWaitRefCount;
Thread->RunningEvents.ThreadSleeping = false;
ctx->IdleWaitCV.notify_all();
}
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext) {
void Dispatcher::StoreThreadState(int Signal, void *ucontext) {
// We can end up getting a signal at any point in our host state
// Jump to a handler that saves all state so we can safely return
uint64_t OldSP = ArchHelpers::Context::GetSp(ucontext);
@@ -65,422 +45,89 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(FEXCore::Core:
// Save guest state
// We can't guarantee if registers are in context or host GPRs
// So we need to save everything
memcpy(&Context->GuestState, Thread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
memcpy(&Context->GuestState, ThreadState->CurrentFrame, sizeof(FEXCore::Core::CPUState));
// Set the new SP
ArchHelpers::Context::SetSp(ucontext, NewSP);
// Signal frames are only used on the interpreter
// The JITS require the stack to be setup correctly on rt_sigreturn
if (CTX->Config.Core() == FEXCore::Config::CONFIG_INTERPRETER) {
SignalFrames.push(NewSP);
}
Context->Flags = 0;
Context->FPStateLocation = 0;
Context->UContextLocation = 0;
Context->SigInfoLocation = 0;
// Store fault to top status and then reset it
Context->FaultToTopAndGeneratedException = Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException;
Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = false;
return Context;
SignalFrames.push(NewSP);
}
void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext) {
uint64_t OldSP{};
if (CTX->Config.Core() == FEXCore::Config::CONFIG_IRJIT) {
OldSP = ArchHelpers::Context::GetSp(ucontext);
}
else {
LOGMAN_THROW_A_FMT(!SignalFrames.empty(), "Trying to restore a signal frame when we don't have any");
OldSP = SignalFrames.top();
SignalFrames.pop();
}
const bool IsAVXEnabled = CTX->Config.EnableAVX;
void Dispatcher::RestoreThreadState(void *ucontext) {
uint64_t OldSP = SignalFrames.top();
SignalFrames.pop();
uintptr_t NewSP = OldSP;
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
// First thing, reset the guest state
memcpy(Thread->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
memcpy(ThreadState->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
// Now restore host state
ArchHelpers::Context::RestoreContext(ucontext, Context);
if (Context->UContextLocation) {
auto Frame = Thread->CurrentFrame;
if (Context->Flags &ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT) {
// XXX: Unsupported since it needs state reconstruction
// If we are in the JIT then SRA might need to be restored to values from the context
// We can't currently support this since it might result in tearing without real state reconstruction
}
if (!(Context->Flags & ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT)) {
auto *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(Context->UContextLocation);
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<siginfo_t*>(Context->SigInfoLocation);
// If the guest modified the RIP then we need to take special precautions here
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] ||
Context->FaultToTopAndGeneratedException) {
// Hack! Go back to the top of the dispatcher top
// This is only safe inside the JIT rather than anything outside of it
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
// Set our state register to point to our guest thread data
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
// XXX: Full context setting
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL];
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
}
Frame->State.flags[1] = 1;
Frame->State.flags[9] = 1;
#define COPY_REG(x) \
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x];
COPY_REG(R8);
COPY_REG(R9);
COPY_REG(R10);
COPY_REG(R11);
COPY_REG(R12);
COPY_REG(R13);
COPY_REG(R14);
COPY_REG(R15);
COPY_REG(RDI);
COPY_REG(RSI);
COPY_REG(RBP);
COPY_REG(RBX);
COPY_REG(RDX);
COPY_REG(RAX);
COPY_REG(RCX);
COPY_REG(RSP);
#undef COPY_REG
auto *xstate = reinterpret_cast<x86_64::xstate*>(guest_uctx->uc_mcontext.fpregs);
auto *fpstate = &xstate->fpstate;
// Copy float registers
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
if (IsAVXEnabled) {
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
memcpy(&Frame->State.xmm.avx.data[i][0], &fpstate->_xmm[i], sizeof(__uint128_t));
}
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
memcpy(&Frame->State.xmm.avx.data[i][2], &xstate->ymmh.ymmh_space[i], sizeof(__uint128_t));
}
} else {
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
}
// FCW store default
Frame->State.FCW = fpstate->fcw;
Frame->State.FTW = fpstate->ftw;
// Deconstruct FSW
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
}
}
else {
auto *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(Context->UContextLocation);
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(Context->SigInfoLocation);
// If the guest modified the RIP then we need to take special precautions here
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] ||
Context->FaultToTopAndGeneratedException) {
// Hack! Go back to the top of the dispatcher top
// This is only safe inside the JIT rather than anything outside of it
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
// Set our state register to point to our guest thread data
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
// XXX: Full context setting
// First 32-bytes of flags is EFLAGS broken out
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL];
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
}
Frame->State.flags[1] = 1;
Frame->State.flags[9] = 1;
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
Frame->State.cs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
Frame->State.ds_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
Frame->State.es_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
Frame->State.fs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
Frame->State.gs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
Frame->State.ss_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
Frame->State.cs_cached = Frame->State.gdt[Frame->State.cs_idx >> 3].base;
Frame->State.ds_cached = Frame->State.gdt[Frame->State.ds_idx >> 3].base;
Frame->State.es_cached = Frame->State.gdt[Frame->State.es_idx >> 3].base;
Frame->State.fs_cached = Frame->State.gdt[Frame->State.fs_idx >> 3].base;
Frame->State.gs_cached = Frame->State.gdt[Frame->State.gs_idx >> 3].base;
Frame->State.ss_cached = Frame->State.gdt[Frame->State.ss_idx >> 3].base;
#define COPY_REG(x) \
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
COPY_REG(RDI);
COPY_REG(RSI);
COPY_REG(RBP);
COPY_REG(RBX);
COPY_REG(RDX);
COPY_REG(RAX);
COPY_REG(RCX);
COPY_REG(RSP);
#undef COPY_REG
auto *xstate = reinterpret_cast<x86::xstate*>(guest_uctx->uc_mcontext.fpregs);
auto *fpstate = &xstate->fpstate;
// Copy float registers
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
}
// Extended XMM state
if (IsAVXEnabled) {
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
}
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
}
} else {
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
}
// FCW store default
Frame->State.FCW = fpstate->fcw;
Frame->State.FTW = fpstate->ftw;
// Deconstruct FSW
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
}
}
}
// Restore the previous signal state
// This allows recursive signals to properly handle signal masking as we are walking back up the list of signals
CTX->SignalDelegation->SetCurrentSignal(Context->Signal);
}
static uint32_t ConvertSignalToTrapNo(int Signal, siginfo_t *HostSigInfo) {
switch (Signal) {
case SIGSEGV:
if (HostSigInfo->si_code == SEGV_MAPERR ||
HostSigInfo->si_code == SEGV_ACCERR) {
// Protection fault
return X86State::X86_TRAPNO_PF;
}
break;
}
// Unknown mapping, fall back to old behaviour and just pass signal
return Signal;
}
static uint32_t ConvertSignalToError(int Signal, siginfo_t *HostSigInfo) {
switch (Signal) {
case SIGSEGV:
if (HostSigInfo->si_code == SEGV_MAPERR ||
HostSigInfo->si_code == SEGV_ACCERR) {
// Protection fault
// Always a user fault for us
// XXX: PF_PROT and PF_WRITE
return X86State::X86_PF_USER;
}
break;
}
// Not a page fault issue
return 0;
}
template <typename T>
static void SetXStateInfo(T* xstate, bool is_avx_enabled) {
auto* fpstate = &xstate->fpstate;
fpstate->sw_reserved.magic1 = x86_64::fpx_sw_bytes::FP_XSTATE_MAGIC;
fpstate->sw_reserved.extended_size = is_avx_enabled ? sizeof(T) : 0;
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_FP |
x86_64::fpx_sw_bytes::FEATURE_SSE;
if (is_avx_enabled) {
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_YMM;
}
fpstate->sw_reserved.xstate_size = fpstate->sw_reserved.extended_size;
if (is_avx_enabled) {
xstate->xstate_hdr.xfeatures = 0;
}
}
bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
auto ContextBackup = StoreThreadState(Thread, Signal, ucontext);
auto Frame = Thread->CurrentFrame;
bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
StoreThreadState(Signal, ucontext);
auto Frame = ThreadState->CurrentFrame;
// Ref count our faults
// We use this to track if it is safe to clear cache
++Thread->CurrentFrame->SignalHandlerRefCounter;
++SignalHandlerRefCounter;
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
// Set the new PC
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
// Set our state register to point to our guest thread data
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
uint64_t OldGuestSP = Frame->State.gregs[X86State::REG_RSP];
uint64_t NewGuestSP = OldGuestSP;
// Pulling from context here
const bool Is64BitMode = CTX->Config.Is64BitMode;
const bool IsAVXEnabled = CTX->Config.EnableAVX;
const uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
// Spill the SRA regardless of signal handler type
// We are going to be returning to the top of the dispatcher which will fill again
// Otherwise we might load garbage
if (config.StaticRegisterAllocation) {
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
uint32_t IgnoreMask{};
#ifdef _M_ARM_64
if (Frame->InSyscallInfo != 0) {
// We are in a syscall, this means we are in a weird register state
// We need to spill SRA but only some of it, since some values have already been spilled
// Lower 16 bits tells us which registers are already spilled to the context
// So we ignore spilling those ones
uint16_t NumRegisters = std::popcount(Frame->InSyscallInfo & 0xFFFF);
if (NumRegisters >= 4) {
// Unhandled case
IgnoreMask = 0;
}
else {
IgnoreMask = Frame->InSyscallInfo & 0xFFFF;
}
}
else {
// We must spill everything
IgnoreMask = 0;
}
#endif
// We are in jit, SRA must be spilled
SpillSRA(Thread, ucontext, IgnoreMask);
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT;
} else {
if (!IsAddressInDispatcher(OldPC)) {
// This is likely to cause issues but in some cases it isn't fatal
// This can also happen if we have put a signal on hold, then we just reenabled the signal
// So we are in the syscall handler
// Only throw a log message in this case
if constexpr (false) {
// XXX: Messages in the signal handler can cause us to crash
LogMan::Msg::EFmt("Signals in dispatcher have unsynchronized context");
}
}
if (!(GuestStack->ss_flags & SS_DISABLE)) {
// If our guest is already inside of the alternative stack
// Then that means we are hitting recursive signals and we need to walk back the stack correctly
uint64_t AltStackBase = reinterpret_cast<uint64_t>(GuestStack->ss_sp);
uint64_t AltStackEnd = AltStackBase + GuestStack->ss_size;
if (OldGuestSP >= AltStackBase &&
OldGuestSP <= AltStackEnd) {
// We are already in the alt stack, the rest of the code will handle adjusting this
}
else {
NewGuestSP = AltStackEnd;
}
}
// altstack is only used if the signal handler was setup with SA_ONSTACK
if (GuestAction->sa_flags & SA_ONSTACK) {
// Additionally the altstack is only used if the enabled (SS_DISABLE flag is not set)
if (!(GuestStack->ss_flags & SS_DISABLE)) {
// If our guest is already inside of the alternative stack
// Then that means we are hitting recursive signals and we need to walk back the stack correctly
uint64_t AltStackBase = reinterpret_cast<uint64_t>(GuestStack->ss_sp);
uint64_t AltStackEnd = AltStackBase + GuestStack->ss_size;
if (OldGuestSP >= AltStackBase &&
OldGuestSP <= AltStackEnd) {
// We are already in the alt stack, the rest of the code will handle adjusting this
}
else {
NewGuestSP = AltStackEnd;
}
}
}
if (Is64BitMode) {
// Back up past the redzone, which is 128bytes
// 32-bit doesn't have a redzone
NewGuestSP -= 128;
}
// siginfo_t
siginfo_t *HostSigInfo = reinterpret_cast<siginfo_t*>(info);
// Backup where we think the RIP currently is
ContextBackup->OriginalRIP = Frame->State.rip;
// Back up past the redzone, which is 128bytes
// Don't need this offset if we aren't going to be putting siginfo in to it
NewGuestSP -= 128;
if (GuestAction->sa_flags & SA_SIGINFO) {
// Setup ucontext a bit
if (Is64BitMode) {
if (IsAVXEnabled) {
NewGuestSP -= sizeof(x86_64::xstate);
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::xstate));
if (SRAEnabled) {
if (!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
LogMan::Throw::A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
} else {
NewGuestSP -= sizeof(x86_64::_libc_fpstate);
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::_libc_fpstate));
// We are in jit, SRA must be spilled
SpillSRA(ucontext);
}
uint64_t FPStateLocation = NewGuestSP;
}
// Setup ucontext a bit
if (CTX->Config.Is64BitMode) {
NewGuestSP -= sizeof(FEXCore::x86_64::ucontext_t);
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::ucontext_t));
uint64_t UContextLocation = NewGuestSP;
NewGuestSP -= sizeof(siginfo_t);
NewGuestSP = AlignDown(NewGuestSP, alignof(siginfo_t));
uint64_t SigInfoLocation = NewGuestSP;
ContextBackup->FPStateLocation = FPStateLocation;
ContextBackup->UContextLocation = UContextLocation;
ContextBackup->SigInfoLocation = SigInfoLocation;
FEXCore::x86_64::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(UContextLocation);
siginfo_t *guest_siginfo = reinterpret_cast<siginfo_t*>(SigInfoLocation);
// We have extended float information
guest_uctx->uc_flags = FEXCore::x86_64::UC_FP_XSTATE;
guest_uctx->uc_flags |= FEXCore::x86_64::UC_FP_XSTATE;
// Pointer to where the fpreg memory is
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<x86_64::_libc_fpstate*>(FPStateLocation);
auto *xstate = reinterpret_cast<x86_64::xstate*>(FPStateLocation);
SetXStateInfo(xstate, IsAVXEnabled);
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = Frame->State.rip;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CSGSFS] = 0;
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
// SI_USER could also potentially have random data in it, needs to be bit perfect
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
*guest_siginfo = *HostSigInfo;
if (ContextBackup->FaultToTopAndGeneratedException) {
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
// Overwrite si_code
guest_siginfo->si_code = Thread->CurrentFrame->SynchronousFaultData.si_code;
Signal = Frame->SynchronousFaultData.Signal;
}
else {
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
}
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_OLDMASK] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CR2] = 0;
guest_uctx->uc_mcontext.fpregs = &guest_uctx->__fpregs_mem;
#define COPY_REG(x) \
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
@@ -502,28 +149,15 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
COPY_REG(RSP);
#undef COPY_REG
auto* fpstate = &xstate->fpstate;
// Copy float registers
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
if (IsAVXEnabled) {
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
}
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
}
} else {
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
}
memcpy(guest_uctx->__fpregs_mem._st, Frame->State.mm, sizeof(Frame->State.mm));
memcpy(guest_uctx->__fpregs_mem._xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
// FCW store default
fpstate->fcw = Frame->State.FCW;
fpstate->ftw = Frame->State.FTW;
guest_uctx->__fpregs_mem.fcw = Frame->State.FCW;
// Reconstruct FSW
fpstate->fsw =
guest_uctx->__fpregs_mem.fsw =
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
@@ -535,292 +169,136 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
guest_uctx->uc_stack.ss_sp = GuestStack->ss_sp;
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
Frame->State.gregs[X86State::REG_RSI] = SigInfoLocation;
// XXX: siginfo_t(RSI)
Frame->State.gregs[X86State::REG_RSI] = 0x4142434445460000;
Frame->State.gregs[X86State::REG_RDX] = UContextLocation;
}
else {
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT;
if (IsAVXEnabled) {
NewGuestSP -= sizeof(x86::xstate);
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::xstate));
} else {
NewGuestSP -= sizeof(x86::_libc_fpstate);
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::_libc_fpstate));
}
uint64_t FPStateLocation = NewGuestSP;
// XXX: 32bit Support
NewGuestSP -= sizeof(FEXCore::x86::ucontext_t);
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::ucontext_t));
uint64_t UContextLocation = NewGuestSP;
uint64_t UContextLocation = 0; // NewGuestSP;
NewGuestSP -= sizeof(FEXCore::x86::siginfo_t);
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::siginfo_t));
uint64_t SigInfoLocation = NewGuestSP;
ContextBackup->FPStateLocation = FPStateLocation;
ContextBackup->UContextLocation = UContextLocation;
ContextBackup->SigInfoLocation = SigInfoLocation;
FEXCore::x86::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(UContextLocation);
FEXCore::x86::siginfo_t *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(SigInfoLocation);
// We have extended float information
guest_uctx->uc_flags = FEXCore::x86::UC_FP_XSTATE;
// Pointer to where the fpreg memory is
guest_uctx->uc_mcontext.fpregs = static_cast<uint32_t>(FPStateLocation);
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
SetXStateInfo(xstate, IsAVXEnabled);
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss_idx;
if (ContextBackup->FaultToTopAndGeneratedException) {
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
Signal = Frame->SynchronousFaultData.Signal;
}
else {
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
guest_siginfo->si_code = HostSigInfo->si_code;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
}
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = 0;
#define COPY_REG(x) \
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
COPY_REG(RDI);
COPY_REG(RSI);
COPY_REG(RBP);
COPY_REG(RBX);
COPY_REG(RDX);
COPY_REG(RAX);
COPY_REG(RCX);
COPY_REG(RSP);
#undef COPY_REG
auto *fpstate = &xstate->fpstate;
// Copy float registers
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
}
// Extended XMM state
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
if (IsAVXEnabled) {
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
}
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
}
} else {
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
}
// FCW store default
fpstate->fcw = Frame->State.FCW;
fpstate->ftw = Frame->State.FTW;
// Reconstruct FSW
fpstate->fsw =
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] << 10) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] << 14);
// Copy over signal stack information
guest_uctx->uc_stack.ss_flags = GuestStack->ss_flags;
guest_uctx->uc_stack.ss_sp = static_cast<uint32_t>(reinterpret_cast<uint64_t>(GuestStack->ss_sp));
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
// These three elements are in every siginfo
guest_siginfo->si_signo = HostSigInfo->si_signo;
guest_siginfo->si_errno = HostSigInfo->si_errno;
switch (Signal) {
case SIGSEGV:
case SIGBUS:
// Macro expansion to get the si_addr
// This is the address trying to be accessed, not the RIP
guest_siginfo->_sifields._sigfault.addr = static_cast<uint32_t>(reinterpret_cast<uintptr_t>(HostSigInfo->si_addr));
break;
case SIGFPE:
case SIGILL:
// Macro expansion to get the si_addr
// Can't really give a real result here. Pull from the context for now
guest_siginfo->_sifields._sigfault.addr = Frame->State.rip;
break;
case SIGCHLD:
guest_siginfo->_sifields._sigchld.pid = HostSigInfo->si_pid;
guest_siginfo->_sifields._sigchld.uid = HostSigInfo->si_uid;
guest_siginfo->_sifields._sigchld.status = HostSigInfo->si_status;
guest_siginfo->_sifields._sigchld.utime = HostSigInfo->si_utime;
guest_siginfo->_sifields._sigchld.stime = HostSigInfo->si_stime;
break;
case SIGALRM:
case SIGVTALRM:
guest_siginfo->_sifields._timer.tid = HostSigInfo->si_timerid;
guest_siginfo->_sifields._timer.overrun = HostSigInfo->si_overrun;
guest_siginfo->_sifields._timer.sigval.sival_int = HostSigInfo->si_int;
break;
default:
LogMan::Msg::EFmt("Unhandled siginfo_t for signal: {}\n", Signal);
break;
}
uint64_t SigInfoLocation = 0; // NewGuestSP;
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = UContextLocation;
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = SigInfoLocation;
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = Signal;
}
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.sigaction);
}
else {
if (!Is64BitMode) {
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = Signal;
}
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.handler);
}
if (Is64BitMode) {
Frame->State.gregs[FEXCore::X86State::REG_RDI] = Signal;
if (CTX->Config.Is64BitMode) {
Frame->State.gregs[X86State::REG_RDI] = Signal;
// Set up the new SP for stack handling
NewGuestSP -= 8;
*(uint64_t*)NewGuestSP = SignalReturn;
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
*(uint64_t*)NewGuestSP = CTX->X86CodeGen.SignalReturn;
Frame->State.gregs[X86State::REG_RSP] = NewGuestSP;
}
else {
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = SignalReturn;
LOGMAN_THROW_AA_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
*(uint32_t*)NewGuestSP = CTX->X86CodeGen.SignalReturn;
LogMan::Throw::A(CTX->X86CodeGen.SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
Frame->State.gregs[X86State::REG_RSP] = NewGuestSP;
}
// The guest starts its signal frame with a zero initialized FPU
// Set that up now. Little bit costly but it's a requirement
// This state will be restored on rt_sigreturn
memset(Frame->State.xmm.avx.data, 0, sizeof(Frame->State.xmm));
memset(Frame->State.mm, 0, sizeof(Frame->State.mm));
Frame->State.FCW = 0x37F;
Frame->State.FTW = 0xFFFF;
return true;
}
bool Dispatcher::HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
bool Dispatcher::HandleSIGILL(int Signal, void *info, void *ucontext) {
if (ArchHelpers::Context::GetPc(ucontext) == SignalHandlerReturnAddress) {
RestoreThreadState(Thread, ucontext);
RestoreThreadState(ucontext);
// Ref count our faults
// We use this to track if it is safe to clear cache
--Thread->CurrentFrame->SignalHandlerRefCounter;
--SignalHandlerRefCounter;
return true;
}
if (ArchHelpers::Context::GetPc(ucontext) == PauseReturnInstruction) {
RestoreThreadState(Thread, ucontext);
RestoreThreadState(ucontext);
// Ref count our faults
// We use this to track if it is safe to clear cache
--Thread->CurrentFrame->SignalHandlerRefCounter;
--SignalHandlerRefCounter;
return true;
}
return false;
}
bool Dispatcher::HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
FEXCore::Core::SignalEvent SignalReason = Thread->SignalReason.load();
auto Frame = Thread->CurrentFrame;
bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
FEXCore::Core::SignalEvent SignalReason = ThreadState->SignalReason.load();
auto Frame = ThreadState->CurrentFrame;
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_PAUSE) {
// Store our thread state so we can come back to this
StoreThreadState(Thread, Signal, ucontext);
StoreThreadState(Signal, ucontext);
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
// We are in jit, SRA must be spilled
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddressSpillSRA);
} else {
if (config.StaticRegisterAllocation) {
if (SRAEnabled) {
// We are in non-jit, SRA is already spilled
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
"Signals in dispatcher have unsynchronized context");
LogMan::Throw::A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
}
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
}
// Set the new PC
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
// Set our state register to point to our guest thread data
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
// Ref count our faults
// We use this to track if it is safe to clear cache
++Thread->CurrentFrame->SignalHandlerRefCounter;
++SignalHandlerRefCounter;
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
return true;
}
if (SignalReason == FEXCore::Core::SignalEvent::Stop) {
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_STOP) {
// Our thread is stopping
// We don't care about anything at this point
// Set the stack to our starting location when we entered the core and get out safely
ArchHelpers::Context::SetSp(ucontext, Frame->ReturningStackLocation);
// Our ref counting doesn't matter anymore
Thread->CurrentFrame->SignalHandlerRefCounter = 0;
SignalHandlerRefCounter = 0;
// Set the new PC
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
// We are in jit, SRA must be spilled
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddressSpillSRA);
} else {
if (config.StaticRegisterAllocation) {
if (SRAEnabled) {
// We are in non-jit, SRA is already spilled
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
"Signals in dispatcher have unsynchronized context");
LogMan::Throw::A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
}
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
}
// We need to be a little bit careful here
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
// Then we need to ensure we don't double decrement our idle thread counter
if (Thread->RunningEvents.ThreadSleeping) {
// If the thread was sleeping then its idle counter was decremented
// Reincrement it here to not break logic
++Thread->CTX->IdleWaitRefCount;
}
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
return true;
}
if (SignalReason == FEXCore::Core::SignalEvent::Return) {
RestoreThreadState(Thread, ucontext);
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_RETURN) {
RestoreThreadState(ucontext);
// Ref count our faults
// We use this to track if it is safe to clear cache
--Thread->CurrentFrame->SignalHandlerRefCounter;
--SignalHandlerRefCounter;
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
return true;
}
@@ -828,15 +306,38 @@ bool Dispatcher::HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, i
}
uint64_t Dispatcher::GetCompileBlockPtr() {
using ClassPtrType = void (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
PtrCast CompileBlockPtr;
CompileBlockPtr.ClassPtr = &FEXCore::Context::Context::CompileBlockJit;
CompileBlockPtr.ClassPtr = &FEXCore::Context::Context::CompileBlock;
return CompileBlockPtr.Data;
}
void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
for (auto iter = CodeBuffers.begin(); iter != CodeBuffers.end(); ++iter) {
auto [start, end] = *iter;
if (start == reinterpret_cast<uint64_t>(start_to_remove)) {
CodeBuffers.erase(iter);
return;
}
}
}
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher) {
for (auto [start, end] : CodeBuffers) {
if (Address >= start && Address < end) {
return true;
}
}
if (IncludeDispatcher) {
return IsAddressInDispatcher(Address);
}
return false;
}
}
@@ -1,37 +1,26 @@
#pragma once
#include <FEXCore/Core/CPUBackend.h>
#include "Interface/Core/ArchHelpers/MContext.h"
#include <FEXCore/Core/SignalDelegator.h>
#include "Interface/Context/Context.h"
#include <cstdint>
#include <signal.h>
#include <stddef.h>
#include <stack>
#include <tuple>
#include <vector>
namespace FEXCore {
struct GuestSigAction;
}
namespace FEXCore::Core {
struct CpuStateFrame;
struct InternalThreadState;
}
namespace FEXCore::Context {
struct Context;
}
namespace FEXCore::CPU {
struct DispatcherConfig {
bool StaticRegisterAllocation = false;
bool ExecuteBlocksWithCall = false;
uintptr_t ExitFunctionLink = 0;
uintptr_t ExitFunctionLinkThis = 0;
bool StaticRegisterAssignment = false;
};
class Dispatcher {
public:
virtual ~Dispatcher() = default;
CPUBackend::AsmDispatch DispatchPtr;
CPUBackend::JITCallback CallbackPtr;
FEXCore::Context::Context::IntCallbackReturn ReturnPtr;
/**
* @name Dispatch Helper functions
@@ -44,70 +33,52 @@ public:
uint64_t ThreadPauseHandlerAddressSpillSRA{};
uint64_t ExitFunctionLinkerAddress{};
uint64_t SignalHandlerReturnAddress{};
uint64_t GuestSignal_SIGILL{};
uint64_t GuestSignal_SIGTRAP{};
uint64_t GuestSignal_SIGSEGV{};
uint64_t IntCallbackReturnAddress{};
uint64_t PauseReturnInstruction{};
/** @} */
uint32_t SignalHandlerRefCounter{};
uint64_t Start{};
uint64_t End{};
bool HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
bool HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
bool HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
bool HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
bool HandleSIGILL(int Signal, void *info, void *ucontext);
bool HandleSignalPause(int Signal, void *info, void *ucontext);
bool IsAddressInDispatcher(uint64_t Address) const {
void RegisterCodeBuffer(uint8_t* start, size_t size) {
CodeBuffers.emplace_back(reinterpret_cast<uint64_t>(start),
reinterpret_cast<uint64_t>(start + size));
}
void RemoveCodeBuffer(uint8_t* start);
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true);
bool IsAddressInDispatcher(uint64_t Address) {
return Address >= Start && Address < End;
}
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
// These are across all arches for now
static constexpr size_t MaxGDBPauseCheckSize = 128;
static constexpr size_t MaxInterpreterTrampolineSize = 128;
virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0;
virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0;
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
virtual void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
DispatchPtr(Frame);
}
virtual void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
CallbackPtr(Frame, RIP);
}
protected:
Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &Config)
Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
: CTX {ctx}
, config {Config}
{}
, ThreadState {Thread} {}
ArchHelpers::Context::ContextBackup* StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext);
void RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext);
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
void StoreThreadState(int Signal, void *ucontext);
void RestoreThreadState(void *ucontext);
std::stack<uint64_t> SignalFrames;
virtual void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {}
bool SRAEnabled = false;
virtual void SpillSRA(void *ucontext) {}
FEXCore::Context::Context *CTX;
DispatcherConfig config;
FEXCore::Core::InternalThreadState *ThreadState;
static void SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuStateFrame *Frame);
static uint64_t GetCompileBlockPtr();
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
AsmDispatch DispatchPtr;
JITCallback CallbackPtr;
private:
std::vector<std::tuple<uint64_t, uint64_t>> CodeBuffers; // Start, End
};
}
}
@@ -1,43 +1,22 @@
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Context/Context.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <cmath>
#include <memory>
#include <stddef.h>
#include <stdint.h>
#include <sys/mman.h>
#include <xbyak/xbyak.h>
#define STATE_PTR(STATE_TYPE, FIELD) \
[STATE + offsetof(FEXCore::Core::STATE_TYPE, FIELD)]
namespace FEXCore::CPU {
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
#define STATE r14
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
: Dispatcher(ctx, config)
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
nullptr) {
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "X86 dispatcher does not support SRA");
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
: Dispatcher(ctx, Thread)
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE) {
using namespace Xbyak;
using namespace Xbyak::util;
DispatchPtr = getCurr<AsmDispatch>();
DispatchPtr = getCurr<CPUBackend::AsmDispatch>();
// Temp registers
// rax, rcx, rdx, rsi, r8, r9,
@@ -83,7 +62,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
// regardless of where we were in the stack
mov(qword STATE_PTR(CpuStateFrame, ReturningStackLocation), rsp);
mov(qword [rdi + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)], rsp);
Label LoopTop;
Label FullLookup;
@@ -96,26 +75,27 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
{
// Load our RIP
mov(rdx, qword STATE_PTR(CPUState, rip));
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
// L1 Cache
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
mov(rax, rdx);
if (!config.ExecuteBlocksWithCall)
{
// L1 Cache
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(rax, rdx);
and_(rax, LookupCache::L1_ENTRIES_MASK);
shl(rax, 4);
cmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)], rdx);
jne(FullLookup);
jmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)]);
and_(rax, LookupCache::L1_ENTRIES_MASK);
shl(rax, 4);
cmp(qword[r13 + rax + 8], rdx);
jne(FullLookup);
jmp(qword[r13 + rax + 0]);
}
L(FullLookup);
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
mov(r13, Thread->LookupCache->GetPagePointer());
// Full lookup
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
mov(rax, rdx);
mov(rbx, VirtualMemorySize - 1);
mov(rbx, Thread->LookupCache->GetVirtualMemorySize() - 1);
and_(rax, rbx);
shr(rax, 12);
@@ -142,15 +122,39 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
je(NoBlock);
// Update L1
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
mov(rcx, rdx);
and_(rcx, LookupCache::L1_ENTRIES_MASK);
shl(rcx, 1);
mov(qword[r13 + rcx*8 + 8], rdx);
mov(qword[r13 + rcx*8 + 0], rax);
if (config.ExecuteBlocksWithCall) {
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(rcx, rdx);
and_(rcx, LookupCache::L1_ENTRIES_MASK);
shl(rcx, 1);
mov(qword[r13 + rcx*8 + 8], rdx);
mov(qword[r13 + rcx*8 + 0], rax);
}
// Real block if we made it here
jmp(rax);
if (!config.ExecuteBlocksWithCall) {
jmp(rax);
} else {
mov(rdi, STATE);
call(rax);
if (CTX->GetGdbServerStatus()) {
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
mov(rax, qword [STATE + (offsetof(FEXCore::Core::InternalThreadState, CTX))]);
// If the value == 0 then branch to the top
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
je(LoopTop);
// Else we need to pause now
jmp(ThreadPauseHandler);
ud2();
}
else {
jmp(LoopTop);
}
}
}
{
@@ -169,122 +173,40 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
ret();
}
constexpr bool SignalSafeCompile = true;
// Block creation
{
L(NoBlock);
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// RDI: SETMASK
// RSI: Pointer to mask value (uint64_t)
// RDX: Pointer to old mask value (uint64_t)
// R10: Size of mask, sizeof(uint64_t)
// RAX: Syscall
using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
// Backup rdx
mov(r9, rdx);
mov(rdi, ~0ULL);
sub(rsp, 16);
mov(qword [rsp], rdi);
mov(qword [rsp + 8], rdi);
mov(rdi, SIG_SETMASK);
mov(rsi, rsp);
mov(rdx, rsp);
mov(r10, 8);
mov(rax, SYS_rt_sigprocmask);
syscall();
mov(rdx, r9);
}
PtrCast Ptr;
Ptr.ClassPtr = &FEXCore::Context::Context::CompileBlock;
// {rdi, rsi, rdx}
mov(rdi, reinterpret_cast<uint64_t>(CTX));
mov(rsi, STATE);
mov(rax, GetCompileBlockPtr());
mov(rax, Ptr.Data);
call(rax);
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
// Backup rdx
mov(r9, rdx);
mov(rdi, SIG_SETMASK);
mov(rsi, rsp);
mov(rdx, 0); // Don't care about result
mov(r10, 8);
mov(rax, SYS_rt_sigprocmask);
syscall();
// Bring stack back
add(rsp, 16);
mov(rdx, r9);
}
// rdx already contains RIP here
jmp(LoopTop);
}
{
ExitFunctionLinkerAddress = getCurr<uint64_t>();
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// RDI: SETMASK
// RSI: Pointer to mask value (uint64_t)
// RDX: Pointer to old mask value (uint64_t)
// R10: Size of mask, sizeof(uint64_t)
// RAX: Syscall
// {rdi, rsi, rdx}
mov(rdi, config.ExitFunctionLinkThis);
mov(rsi, STATE);
mov(rdx, rax); // rax is set at the block end
// Backup rax
mov(r9, rax);
mov(rdi, ~0ULL);
sub(rsp, 16);
mov(qword [rsp], rdi);
mov(qword [rsp + 8], rdi);
mov(rdi, SIG_SETMASK);
mov(rsi, rsp);
mov(rdx, rsp);
mov(r10, 8);
mov(rax, SYS_rt_sigprocmask);
syscall();
mov(rax, r9);
}
// {rdi, rsi}
mov(rdi, STATE);
mov(rsi, rax); // rax is set at the block end
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
// Backup rax
mov(r9, rax);
mov(rdi, SIG_SETMASK);
mov(rsi, rsp);
mov(rdx, 0); // Don't care about result
mov(r10, 8);
mov(rax, SYS_rt_sigprocmask);
syscall();
// Bring stack back
add(rsp, 16);
jmp(r9);
}
else {
jmp(rax);
}
mov(rax, config.ExitFunctionLink);
call(rax);
jmp(rax);
}
{
@@ -304,7 +226,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
}
{
CallbackPtr = getCurr<JITCallback>();
CallbackPtr = getCurr<CPUBackend::JITCallback>();
push(rbx);
push(rbp);
@@ -319,7 +241,8 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
// XXX: XMM?
// Make sure to adjust the refcounter so we don't clear the cache now
add(qword STATE_PTR(CpuStateFrame, SignalHandlerRefCounter), 1);
mov(rax, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
add(dword [rax], 1);
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
@@ -327,12 +250,12 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
// Store the trampoline to the guest stack
// Guest stack is now correctly misaligned after a regular call instruction
sub(qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]), 16);
mov(rbx, qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 16);
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])]);
mov(qword [rbx], rax);
// Store RIP to the context state
mov(qword STATE_PTR(CpuStateFrame, State.rip), rsi);
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], rsi);
// Back to the loop top now
jmp(LoopTop);
@@ -341,41 +264,14 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
{
// Signal return handler
SignalHandlerReturnAddress = getCurr<uint64_t>();
ud2();
}
{
// Guest SIGILL handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGILL = getCurr<uint64_t>();
ud2();
}
{
// Guest SIGTRAP handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGTRAP = getCurr<uint64_t>();
// ud2 = SIGILL
// int3 = SIGTRAP
// hlt = SIGSEGV
int3();
}
{
// Guest SIGSEGV handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGSEGV = getCurr<uint64_t>();
// ud2 = SIGILL
// int3 = SIGTRAP
// hlt = SIGSEGV
hlt();
}
{
IntCallbackReturnAddress = getCurr<uint64_t>();
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
// using CallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
// rdi = thread
// rsi = rsp
@@ -400,100 +296,30 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
Start = reinterpret_cast<uint64_t>(getCode());
End = Start + getSize();
if (CTX->Config.BlockJITNaming()) {
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
CTX->Symbols.Register(reinterpret_cast<void*>(Start), End-Start, Name);
}
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
}
}
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline
static thread_local Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
using namespace Xbyak;
using namespace Xbyak::util;
emit.setNewBuffer(CodeBuffer, MaxGDBPauseCheckSize);
Label RunBlock;
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
emit.mov(rax, reinterpret_cast<uint64_t>(CTX));
// If the value == 0 then we don't need to stop
emit.cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
emit.je(RunBlock);
{
// Make sure RIP is syncronized to the context
emit.mov(rax, GuestRIP);
emit.mov(qword STATE_PTR(CpuStateFrame, State.rip), rax);
// Stop the thread
emit.mov(rax, qword STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
emit.jmp(rax);
}
emit.L(RunBlock);
emit.ready();
return emit.getSize();
}
size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
using namespace Xbyak;
using namespace Xbyak::util;
emit.setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
Label InlineIRData;
emit.mov(rdi, STATE);
emit.lea(rsi, ptr[rip + InlineIRData]);
emit.call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
emit.jmp(qword STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
emit.L(InlineIRData);
emit.ready();
return emit.getSize();
#if ENABLE_JITSYMBOLS
std::string Name = "Dispatch_" + std::to_string(::gettid());
CTX->Symbols.Register(Start, End-Start, Name);
#endif
}
X86Dispatcher::~X86Dispatcher() {
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
}
#ifdef _M_X86_64
void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto &Common = Thread->CurrentFrame->Pointers.Common;
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
DispatcherConfig config;
config.ExecuteBlocksWithCall = true;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddress;
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddress;
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
Common.SignalReturnHandler = SignalHandlerReturnAddress;
Dispatcher = new X86Dispatcher(ctx, Thread, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
(uintptr_t&)Interpreter.CallbackReturn = IntCallbackReturnAddress;
}
// TODO: It feels wrong to initialize this way
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
}
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
return std::make_unique<X86Dispatcher>(CTX, Config);
}
#endif
}
@@ -5,24 +5,13 @@
#define XBYAK64
#include <xbyak/xbyak.h>
namespace FEXCore::Context {
struct Context;
}
namespace FEXCore::Core {
struct InternalThreadState;
}
namespace FEXCore::CPU {
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
public:
X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
virtual ~X86Dispatcher() override;
};
}
}
+139 -330
View File
@@ -7,32 +7,20 @@ $end_info$
#include "Interface/Context/Context.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/InternalThreadState.h"
#include <array>
#include <assert.h>
#include <algorithm>
#include <cstring>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/X86Tables.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <set>
#include <sys/mman.h>
namespace FEXCore::Frontend {
#include "Interface/Core/VSyscall/VSyscall.inc"
using namespace FEXCore::X86Tables;
static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool HasREX, bool HasXMM, bool HasMM, uint8_t InvalidOffset = 16) {
using GPRArray = std::array<uint32_t, 16>;
static constexpr GPRArray GPRIndexes = {
constexpr std::array<uint64_t, 16> GPRIndexes = {
// Classical ordering?
FEXCore::X86State::REG_RAX,
FEXCore::X86State::REG_RCX,
@@ -52,7 +40,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
FEXCore::X86State::REG_R15,
};
static constexpr GPRArray GPR8BitHighIndexes = {
constexpr std::array<uint64_t, 16> GPR8BitHighIndexes = {
// Classical ordering?
FEXCore::X86State::REG_RAX,
FEXCore::X86State::REG_RCX,
@@ -72,7 +60,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
FEXCore::X86State::REG_R15,
};
static constexpr GPRArray XMMIndexes = {
constexpr std::array<uint64_t, 16> XMMIndexes = {
FEXCore::X86State::REG_XMM_0,
FEXCore::X86State::REG_XMM_1,
FEXCore::X86State::REG_XMM_2,
@@ -91,7 +79,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
FEXCore::X86State::REG_XMM_15,
};
static constexpr GPRArray MMIndexes = {
constexpr std::array<uint64_t, 16> MMIndexes = {
FEXCore::X86State::REG_MM_0,
FEXCore::X86State::REG_MM_1,
FEXCore::X86State::REG_MM_2,
@@ -110,7 +98,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
FEXCore::X86State::REG_INVALID
};
const GPRArray *GPRs = &GPRIndexes;
const std::array<uint64_t, 16> *GPRs = &GPRIndexes;
if (HasXMM) {
GPRs = &XMMIndexes;
}
@@ -129,79 +117,33 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
return (*GPRs)[(REX << 3) | bits];
}
static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
using GPRArray = std::array<uint32_t, 16>;
static constexpr GPRArray GPRIndexes = {
FEXCore::X86State::REG_RAX,
FEXCore::X86State::REG_RCX,
FEXCore::X86State::REG_RDX,
FEXCore::X86State::REG_RBX,
FEXCore::X86State::REG_RSP,
FEXCore::X86State::REG_RBP,
FEXCore::X86State::REG_RSI,
FEXCore::X86State::REG_RDI,
FEXCore::X86State::REG_R8,
FEXCore::X86State::REG_R9,
FEXCore::X86State::REG_R10,
FEXCore::X86State::REG_R11,
FEXCore::X86State::REG_R12,
FEXCore::X86State::REG_R13,
FEXCore::X86State::REG_R14,
FEXCore::X86State::REG_R15,
};
static constexpr GPRArray XMMIndexes = {
FEXCore::X86State::REG_XMM_0,
FEXCore::X86State::REG_XMM_1,
FEXCore::X86State::REG_XMM_2,
FEXCore::X86State::REG_XMM_3,
FEXCore::X86State::REG_XMM_4,
FEXCore::X86State::REG_XMM_5,
FEXCore::X86State::REG_XMM_6,
FEXCore::X86State::REG_XMM_7,
FEXCore::X86State::REG_XMM_8,
FEXCore::X86State::REG_XMM_9,
FEXCore::X86State::REG_XMM_10,
FEXCore::X86State::REG_XMM_11,
FEXCore::X86State::REG_XMM_12,
FEXCore::X86State::REG_XMM_13,
FEXCore::X86State::REG_XMM_14,
FEXCore::X86State::REG_XMM_15,
};
if (HasXMM) {
return XMMIndexes[vvvv];
} else {
return GPRIndexes[vvvv];
}
}
Decoder::Decoder(FEXCore::Context::Context *ctx)
: CTX {ctx}
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN }
, PoolObject {ctx->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
}
Decoder::~Decoder() {
PoolObject.UnclaimBuffer();
: CTX {ctx} {
DecodedBuffer.resize(DefaultDecodedBufferSize);
}
uint8_t Decoder::ReadByte() {
uint8_t Byte = InstStream[InstructionSize];
LOGMAN_THROW_AA_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
LogMan::Throw::A(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
Instruction[InstructionSize] = Byte;
InstructionSize++;
return Byte;
}
uint8_t Decoder::PeekByte(uint8_t Offset) const {
uint8_t Decoder::PeekByte(uint8_t Offset) {
uint8_t Byte = InstStream[InstructionSize + Offset];
return Byte;
}
uint64_t Decoder::ReadData(uint8_t Size) {
LOGMAN_THROW_AA_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
if (Size == 0) {
return 0;
}
if (Size > sizeof(uint64_t)) {
LogMan::Msg::A("Unknown data size to read");
return 0;
}
uint64_t Res = 0;
std::memcpy(&Res, &InstStream[InstructionSize], Size);
@@ -254,9 +196,9 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModR
}
}
Operand->Type = DecodedOperand::OpType::SIB;
Operand->Data.SIB.Scale = 1;
Operand->Data.SIB.Offset = Literal;
Operand->TypeSIB.Type = DecodedOperand::TYPE_SIB;
Operand->TypeSIB.Scale = 1;
Operand->TypeSIB.Offset = Literal;
// Only called when ModRM.mod != 0b11
struct Encodings {
@@ -269,34 +211,34 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModR
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_INVALID, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RSI, 255},
{FEXCore::X86State::REG_RDI, 255},
{255, 255},
{FEXCore::X86State::REG_RBX, 255},
// Mod = 0b01
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RSI, 255},
{FEXCore::X86State::REG_RDI, 255},
{FEXCore::X86State::REG_RBP, 255},
{FEXCore::X86State::REG_RBX, 255},
// Mod = 0b10
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RSI, 255},
{FEXCore::X86State::REG_RDI, 255},
{FEXCore::X86State::REG_RBP, 255},
{FEXCore::X86State::REG_RBX, 255},
}};
uint8_t LookupIndex = ModRM.mod << 3 | ModRM.rm;
auto it = Lookup[LookupIndex];
Operand->Data.SIB.Base = it.Base;
Operand->Data.SIB.Index = it.Index;
Operand->TypeSIB.Base = it.Base;
Operand->TypeSIB.Index = it.Index;
}
void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM) {
@@ -334,79 +276,79 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
}
// SIB
Operand->Type = DecodedOperand::OpType::SIB;
Operand->Data.SIB.Scale = 1 << SIB.scale;
Operand->TypeSIB.Type = DecodedOperand::TYPE_SIB;
Operand->TypeSIB.Scale = 1 << SIB.scale;
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
Operand->TypeSIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
Operand->TypeSIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
uint64_t Literal {0};
LogMan::Throw::A(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
if (Displacement) {
uint64_t Literal = ReadData(Displacement);
if (Displacement == 1) {
Literal = static_cast<int8_t>(Literal);
}
Operand->Data.SIB.Offset = Literal;
Literal = ReadData(Displacement);
if (Displacement == 1) {
Literal = static_cast<int8_t>(Literal);
}
Operand->TypeSIB.Offset = Literal;
}
else if (ModRM.mod == 0) {
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
if (ModRM.rm == 0b101) {
// 32bit Displacement
const uint32_t Literal = ReadData(4);
uint32_t Literal;
Literal = ReadData(4);
Operand->Type = DecodedOperand::OpType::RIPRelative;
Operand->Data.RIPLiteral.Value.u = Literal;
Operand->TypeRIPLiteral.Type = DecodedOperand::TYPE_RIP_RELATIVE;
Operand->TypeRIPLiteral.Literal.u = Literal;
}
else {
// Register-direct addressing
Operand->Type = DecodedOperand::OpType::GPRDirect;
Operand->Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
Operand->TypeGPR.Type = DecodedOperand::TYPE_GPR_DIRECT;
Operand->TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
}
}
else {
uint8_t DisplacementSize = ModRM.mod == 1 ? 1 : 4;
uint32_t Literal = ReadData(DisplacementSize);
uint32_t Literal{};
Literal = ReadData(DisplacementSize);
if (DisplacementSize == 1) {
Literal = static_cast<int8_t>(Literal);
}
Displacement = DisplacementSize;
Operand->Type = DecodedOperand::OpType::GPRIndirect;
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
Operand->Data.GPRIndirect.Displacement = Literal;
Operand->TypeGPRIndirect.Type = DecodedOperand::TYPE_GPR_INDIRECT;
Operand->TypeGPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
Operand->TypeGPRIndirect.Displacement = Literal;
}
}
bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options) {
bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op) {
DecodeInst->OP = Op;
DecodeInst->TableInfo = Info;
// XXX: Once we support 32bit x86 then this will be necessary to support
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
LogMan::Msg::DFmt("Legacy Prefix");
LogMan::Msg::D("Legacy Prefix");
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
return false;
}
LOGMAN_THROW_AA_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
"Group Ops should have been decoded before this!");
LogMan::Throw::A(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
"Group Ops should have been decoded before this!");
uint8_t DestSize{};
const bool HasWideningDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST) != 0 ||
(Options.w && CTX->Config.Is64BitMode);
const bool HasNarrowingDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST) != 0;
bool HasWideningDisplacement = FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST;
bool HasNarrowingDisplacement = FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST;
bool HasXMMSrc = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_GPR) &&
@@ -439,8 +381,8 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
// New instruction size decoding
{
// Decode destinations first
const auto DstSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeDstFlags(Info->Flags);
const auto SrcSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeSrcFlags(Info->Flags);
uint32_t DstSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeDstFlags(Info->Flags);
uint32_t SrcSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeSrcFlags(Info->Flags);
if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_8BIT) {
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_8BIT);
@@ -517,25 +459,22 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RDX)) {
// Some instructions hardcode their destination as RAX
CurrentDest->Type = DecodedOperand::OpType::GPR;
CurrentDest->Data.GPR.HighBits = false;
CurrentDest->Data.GPR.GPR = HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
CurrentDest->TypeGPR.Type = DecodedOperand::TYPE_GPR;
CurrentDest->TypeGPR.HighBits = false;
CurrentDest->TypeGPR.GPR = HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
CurrentDest = &DecodeInst->Src[0];
}
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
LOGMAN_THROW_AA_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
LogMan::Throw::A(!HasMODRM, "This instruction shouldn't have ModRM!");
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
// This also means that the destination is always a GPR on these ones
// ADDITIONALLY:
// If there is a REX prefix then that allows extended GPR usage
CurrentDest->Type = DecodedOperand::OpType::GPR;
DecodeInst->Dest.Data.GPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100) || HasHighXMM;
CurrentDest->Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
if (CurrentDest->Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
return false;
CurrentDest->TypeGPR.Type = DecodedOperand::TYPE_GPR;
DecodeInst->Dest.TypeGPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100) || HasHighXMM;
CurrentDest->TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
}
uint8_t Bytes = Info->MoreBytes;
@@ -558,86 +497,55 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
ModRM.Hex = DecodeInst->ModRM;
// Decode the GPR source first
GPR.Type = DecodedOperand::OpType::GPR;
GPR.Data.GPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX) || HasHighXMM;
GPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_R ? 1 : 0, ModRM.reg, GPR8Bit, HasREX, HasXMMGPR, HasMMGPR);
if (GPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
return false;
GPR.TypeGPR.Type = DecodedOperand::TYPE_GPR;
GPR.TypeGPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX) || HasHighXMM;
GPR.TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_R ? 1 : 0, ModRM.reg, GPR8Bit, HasREX, HasXMMGPR, HasMMGPR);
// ModRM.mod == 0b11 == Register
// ModRM.Mod != 0b11 == Register-direct addressing
if (ModRM.mod == 0b11) {
NonGPR.Type = DecodedOperand::OpType::GPR;
NonGPR.Data.GPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX) || HasHighXMM;
NonGPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, NonGPR8Bit, HasREX, HasXMMNonGPR, HasMMNonGPR);
if (NonGPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
return false;
NonGPR.TypeGPR.Type = DecodedOperand::TYPE_GPR;
NonGPR.TypeGPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX) || HasHighXMM;
NonGPR.TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, NonGPR8Bit, HasREX, HasXMMNonGPR, HasMMNonGPR);
}
else {
// Only decode if we haven't pre-decoded
if (NonGPR.IsNone()) {
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
(this->*Disp)(&NonGPR, ModRM);
}
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
(this->*Disp)(&NonGPR, ModRM);
}
return true;
};
size_t CurrentSrc = 0;
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) != 0) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
++CurrentSrc;
}
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM) {
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SF_MOD_DST) {
if (!ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest))
return false;
ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest);
}
else {
if (!ModRMOperand(DecodeInst->Dest, DecodeInst->Src[CurrentSrc], HasXMMDst, HasXMMSrc, HasMMDst, HasMMSrc, Is8BitDest, Is8BitSrc))
return false;
ModRMOperand(DecodeInst->Dest, DecodeInst->Src[CurrentSrc], HasXMMDst, HasXMMSrc, HasMMDst, HasMMSrc, Is8BitDest, Is8BitSrc);
}
++CurrentSrc;
}
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) != 0) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
++CurrentSrc;
}
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RAX)) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RAX;
DecodeInst->Src[CurrentSrc].TypeGPR.Type = DecodedOperand::TYPE_GPR;
DecodeInst->Src[CurrentSrc].TypeGPR.HighBits = false;
DecodeInst->Src[CurrentSrc].TypeGPR.GPR = FEXCore::X86State::REG_RAX;
++CurrentSrc;
}
else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RCX)) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RCX;
DecodeInst->Src[CurrentSrc].TypeGPR.Type = DecodedOperand::TYPE_GPR;
DecodeInst->Src[CurrentSrc].TypeGPR.HighBits = false;
DecodeInst->Src[CurrentSrc].TypeGPR.GPR = FEXCore::X86State::REG_RCX;
++CurrentSrc;
}
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) != 0) {
CurrentDest->Type = DecodedOperand::OpType::GPR;
CurrentDest->Data.GPR.HighBits = false;
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
}
if (Bytes != 0) {
LOGMAN_THROW_AA_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
LogMan::Throw::A(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
DecodeInst->Src[CurrentSrc].TypeLiteral.Size = Bytes;
uint64_t Literal = ReadData(Bytes);
uint64_t Literal {0};
Literal = ReadData(Bytes);
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT) ||
(DecodeFlags::GetSizeDstFlags(DecodeInst->Flags) == DecodeFlags::SIZE_64BIT && Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT64BIT)) {
@@ -650,16 +558,15 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
else {
Literal = static_cast<int32_t>(Literal);
}
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
DecodeInst->Src[CurrentSrc].TypeLiteral.Size = DestSize;
}
Bytes = 0;
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
DecodeInst->Src[CurrentSrc].TypeLiteral.Type = DecodedOperand::TYPE_LITERAL;
DecodeInst->Src[CurrentSrc].TypeLiteral.Literal = Literal;
}
LOGMAN_THROW_AA_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
LogMan::Throw::A(Bytes == 0, "Inst at 0x%lx: 0x%04x '%s' Had an instruction of size %d with %d remaining", DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name, InstructionSize, Bytes);
DecodeInst->InstSize = InstructionSize;
return true;
}
@@ -670,22 +577,21 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
// XXX: Once we support 32bit x86 then this will be necessary to support
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
LogMan::Msg::DFmt("Legacy Prefix");
LogMan::Msg::D("Legacy Prefix");
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
return false;
}
LOGMAN_THROW_AA_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
"REX PREFIX should have been decoded before this!");
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
@@ -741,7 +647,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
3,
};
uint8_t Field = RegToField[ModRM.reg];
LOGMAN_THROW_AA_FMT(Field != 255, "Invalid field selected!");
LogMan::Throw::A(Field != 255, "Invalid field selected!");
LocalOp = (Field << 3) | ModRM.rm;
return NormalOp(&SecondModRMTableOps[LocalOp], LocalOp);
@@ -763,38 +669,19 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
return NormalOp(&X87Ops[X87Op], X87Op);
}
else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
uint16_t map_select = 1;
uint16_t pp = 0;
const uint8_t Byte1 = ReadByte();
DecodedHeader options{};
if ((Byte1 & 0b10000000) == 0) {
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.R shouldn't be 0 in 32-bit mode!");
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
}
uint8_t Byte1 = ReadByte();
if (Op == 0xC5) { // Two byte VEX
pp = Byte1 & 0b11;
options.vvvv = 15 - ((Byte1 & 0b01111000) >> 3);
}
else { // 0xC4 = Three byte VEX
const uint8_t Byte2 = ReadByte();
uint8_t Byte2 = ReadByte();
pp = Byte2 & 0b11;
map_select = Byte1 & 0b11111;
options.vvvv = 15 - ((Byte2 & 0b01111000) >> 3);
options.w = (Byte2 & 0b10000000) != 0;
if ((Byte1 & 0b01000000) == 0) {
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.X shouldn't be 0 in 32-bit mode!");
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
}
if (CTX->Config.Is64BitMode && (Byte1 & 0b00100000) == 0) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
}
if (!(map_select >= 1 && map_select <= 3)) {
LogMan::Msg::EFmt("We don't understand a map_select of: {}", map_select);
return false;
}
LogMan::Throw::A(map_select >= 1 && map_select <= 3, "We don't understand a map_select of: %d", map_select);
}
uint16_t VEXOp = ReadByte();
@@ -806,7 +693,6 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
if (LocalInfo->Type >= FEXCore::X86Tables::TYPE_VEX_GROUP_12 &&
LocalInfo->Type <= FEXCore::X86Tables::TYPE_VEX_GROUP_17) {
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
@@ -818,14 +704,12 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
Op = OPD(LocalInfo->Type, pp, ModRM.reg);
#undef OPD
return NormalOp(&VEXTableGroupOps[Op], Op, options);
} else {
return NormalOp(LocalInfo, Op, options);
return NormalOp(&VEXTableGroupOps[Op], Op);
}
else
return NormalOp(LocalInfo, Op);
}
else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
/* uint8_t P1 = */ ReadByte();
/* uint8_t P2 = */ ReadByte();
/* uint8_t P3 = */ ReadByte();
@@ -845,14 +729,12 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
DecodeInst->PC = PC;
for(;;) {
if (InstructionSize >= MAX_INST_SIZE)
return false;
uint8_t Op = ReadByte();
switch (Op) {
case 0x0F: {// Escape Op
uint8_t EscapeOp = ReadByte();
switch (EscapeOp) {
case 0x0F: [[unlikely]] { // 3DNow!
case 0x0F: { // 3DNow!
// 3DNow! Instruction Encoding: 0F 0F [ModRM] [SIB] [Displacement] [Opcode]
// Decode ModRM
uint8_t ModRMByte = ReadByte();
@@ -865,12 +747,8 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
const bool Has16BitAddressing = !CTX->Config.Is64BitMode &&
DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
// All 3DNow! instructions have the second argument as the rm handler
// We need to decode it upfront to get the displacement out of the way
if (ModRM.mod != 0b11) {
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
(this->*Disp)(&DecodeInst->Src[0], ModRM);
}
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
(this->*Disp)(&DecodeInst->Src[0], ModRM);
// Take a peek at the op just past the displacement
uint8_t LocalOp = ReadByte();
@@ -879,20 +757,14 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
}
case 0x38: { // F38 Table!
constexpr uint16_t PF_38_NONE = 0;
constexpr uint16_t PF_38_66 = (1U << 0);
constexpr uint16_t PF_38_F2 = (1U << 1);
constexpr uint16_t PF_38_F3 = (1U << 2);
constexpr uint16_t PF_38_66 = 1;
constexpr uint16_t PF_38_F2 = 2;
uint16_t Prefix = PF_38_NONE;
if (DecodeInst->Flags & DecodeFlags::FLAG_OPERAND_SIZE) {
Prefix |= PF_38_66;
}
if (DecodeInst->Flags & DecodeFlags::FLAG_REPNE_PREFIX) {
Prefix |= PF_38_F2;
}
if (DecodeInst->Flags & DecodeFlags::FLAG_REP_PREFIX) {
Prefix |= PF_38_F3;
}
if (DecodeInst->LastEscapePrefix == 0xF2) // REPNE
Prefix = PF_38_F2;
else if (DecodeInst->LastEscapePrefix == 0x66) // Operand Size
Prefix = PF_38_66;
uint16_t LocalOp = (Prefix << 8) | ReadByte();
return NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp);
@@ -1007,7 +879,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
auto Info = &FEXCore::X86Tables::BaseOps[Op];
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
LogMan::Throw::A(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
// Widening displacement
@@ -1037,10 +909,6 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
}
if (DecodeInst->Dest.IsGPR()) {
assert(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID);
}
return true;
}
@@ -1050,7 +918,7 @@ void Decoder::BranchTargetInMultiblockRange() {
// If the RIP setting is conditional AND within our symbol range then it can be considered for multiblock
uint64_t TargetRIP = 0;
const uint8_t GPRSize = CTX->GetGPRSize();
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
bool Conditional = true;
switch (DecodeInst->OP) {
@@ -1060,23 +928,19 @@ void Decoder::BranchTargetInMultiblockRange() {
// auto RIPOffset = LoadSource(Op, Op->Src[0], Op->Flags);
// auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
// Target offset is PC + InstSize + Literal
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
LogMan::Throw::A(DecodeInst->Src[0].TypeNone.Type == DecodedOperand::TYPE_LITERAL, "Had wrong operand type");
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].TypeLiteral.Literal;
break;
}
case 0xE9:
case 0xEB: // Both are unconditional JMP instructions
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
LogMan::Throw::A(DecodeInst->Src[0].TypeNone.Type == DecodedOperand::TYPE_LITERAL, "Had wrong operand type");
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].TypeLiteral.Literal;
Conditional = false;
break;
case 0xE8: // Call - Immediate target, We don't want to inline calls
if (ExternalBranches) {
ExternalBranches->insert(DecodeInst->PC + DecodeInst->InstSize);
}
[[fallthrough]];
case 0xC2: // RET imm
case 0xC3: // RET
case 0xE8: // Call - Immediate target, We don't want to inline calls
default:
return;
break;
@@ -1106,34 +970,10 @@ void Decoder::BranchTargetInMultiblockRange() {
BlocksToDecode.find(TargetRIP) == BlocksToDecode.end()) {
BlocksToDecode.emplace(TargetRIP);
}
} else {
if (ExternalBranches) {
ExternalBranches->insert(TargetRIP);
}
}
}
const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64 &&
RIP >= VSyscall_Base &&
RIP < VSyscall_End) {
// VSyscall
// This doesn't exist on AArch64 and on x86_64 hosts this is emulated with faults to a region mapped with --xp permissions
// Offset 0: vgettimeofday
// Offset 0x400: vtime
// Offset 0x800: vgetcpu
uint64_t Offset = RIP - VSyscall_Base;
return VSyscallData + Offset;
}
return _InstStream - EntryPoint + RIP;
}
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC) {
Blocks.clear();
BlocksToDecode.clear();
HasBlocks.clear();
@@ -1141,19 +981,19 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
DecodedSize = 0;
MaxCondBranchForward = 0;
MaxCondBranchBackwards = ~0ULL;
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
// XXX: Load symbol data
SymbolAvailable = false;
EntryPoint = PC;
InstStream = _InstStream;
bool ErrorDuringDecoding = false;
uint64_t TotalInstructions{};
// If we don't have symbols available then we become a bit optimistic about multiblock ranges
if (!SymbolAvailable) {
// If we don't have a symbol available then assume all branches are valid for multiblock
SymbolMaxAddress = SectionMaxAddress;
SymbolMaxAddress = ~0ULL;
SymbolMinAddress = EntryPoint;
}
@@ -1163,12 +1003,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
// Entry is a jump target
BlocksToDecode.emplace(PC);
uint64_t CurrentCodePage = PC & FHU::FEX_PAGE_MASK;
std::set<uint64_t> CodePages = { CurrentCodePage };
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
while (!BlocksToDecode.empty()) {
auto BlockDecodeIt = BlocksToDecode.begin();
uint64_t RIPToDecode = *BlockDecodeIt;
@@ -1182,41 +1016,20 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
uint64_t BlockStartOffset = DecodedSize;
// Do a bit of pointer math to figure out where we are in code
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
InstStream = _InstStream - EntryPoint + RIPToDecode;
while (1) {
// MAX_INST_SIZE assumes worst case
auto OpMinAddress = RIPToDecode + PCOffset;
auto OpMaxAddress = OpMinAddress + MAX_INST_SIZE;
auto OpMinPage = OpMinAddress & FHU::FEX_PAGE_MASK;
auto OpMaxPage = OpMaxAddress & FHU::FEX_PAGE_MASK;
if (OpMinPage != CurrentCodePage) {
CurrentCodePage = OpMinPage;
if (CodePages.insert(CurrentCodePage).second) {
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
}
}
if (OpMaxPage != CurrentCodePage) {
CurrentCodePage = OpMaxPage;
if (CodePages.insert(CurrentCodePage).second) {
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
}
}
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
if (ErrorDuringDecoding) {
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", RIPToDecode + PCOffset, PC);
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
LogMan::Msg::D("Couldn't Decode something at 0x%lx, Started at 0x%lx", PC + PCOffset, PC);
LogMan::Throw::A(Blocks.size() != 1, "Decode Error in entry block");
CurrentBlockDecoding.HasInvalidInstruction = true;
// Error while decoding instruction. We don't know the table or instruction size
DecodeInst->TableInfo = nullptr;
DecodeInst->InstSize = 0;
if (ErrorDuringDecoding && Blocks.size() != 1) {
ErrorDuringDecoding = false;
}
break;
}
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
@@ -1225,11 +1038,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
++BlockNumberOfInstructions;
++DecodedSize;
// Can not continue this block at all on invalid instruction
if (CurrentBlockDecoding.HasInvalidInstruction) {
break;
}
bool CanContinue = false;
if (!(DecodeInst->TableInfo->Flags &
(FEXCore::X86Tables::InstFlags::FLAGS_BLOCK_END | FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP))) {
@@ -1249,7 +1057,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
}
if (DecodedSize >= CTX->Config.MaxInstPerBlock ||
DecodedSize >= DefaultDecodedBufferSize) {
DecodedSize >= DecodedBuffer.size()) {
break;
}
@@ -1266,7 +1074,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
// Copy over only the number of instructions we decoded
CurrentBlockDecoding.NumInstructions = BlockNumberOfInstructions;
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer[BlockStartOffset];
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer.at(BlockStartOffset);
}
@@ -1274,6 +1082,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
std::sort(Blocks.begin(), Blocks.end(), [](const FEXCore::Frontend::Decoder::DecodedBlocks& a, const FEXCore::Frontend::Decoder::DecodedBlocks& b) {
return a.Entry < b.Entry;
});
return !ErrorDuringDecoding;
}
}
+9 -36
View File
@@ -1,13 +1,11 @@
#pragma once
#include <FEXCore/Debug/X86Tables.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <array>
#include <cstdint>
#include <utility>
#include <set>
#include <stddef.h>
#include <stack>
#include <vector>
namespace FEXCore::Context {
@@ -26,49 +24,31 @@ public:
};
Decoder(FEXCore::Context::Context *ctx);
~Decoder();
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
bool DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
std::vector<DecodedBlocks> const *GetDecodedBlocks() {
return &Blocks;
}
uint64_t DecodedMinAddress {};
uint64_t DecodedMaxAddress {~0ULL};
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
void DelayedDisownBuffer() {
PoolObject.DelayedDisownBuffer();
}
private:
// To pass any information from instruction prefixes
// down into the actual instruction handling machinery.
struct DecodedHeader {
uint8_t vvvv; // Encoded operand in a VEX prefix.
bool w; // VEX.W bit.
};
FEXCore::Context::Context *CTX;
const FEXCore::HLE::SyscallOSABI OSABI{};
bool DecodeInstruction(uint64_t PC);
void BranchTargetInMultiblockRange();
uint8_t ReadByte();
uint8_t PeekByte(uint8_t Offset) const;
uint8_t PeekByte(uint8_t Offset);
uint64_t ReadData(uint8_t Size);
void SkipBytes(uint8_t Size) { InstructionSize += Size; }
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
bool NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
std::vector<FEXCore::X86Tables::DecodedInst> DecodedBuffer;
size_t DecodedSize {};
uint8_t const *InstStream;
@@ -85,26 +65,19 @@ private:
uint64_t MaxCondBranchBackwards {~0ULL};
uint64_t SymbolMaxAddress {};
uint64_t SymbolMinAddress {~0ULL};
uint64_t SectionMaxAddress {~0ULL};
std::vector<DecodedBlocks> Blocks;
std::set<uint64_t> BlocksToDecode;
std::set<uint64_t> HasBlocks;
std::set<uint64_t> *ExternalBranches {nullptr};
// ModRM rm decoding
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp{
const std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp {
&FEXCore::Frontend::Decoder::DecodeModRM_64,
&FEXCore::Frontend::Decoder::DecodeModRM_16,
};
const uint8_t *AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP);
FEXCORE_TELEMETRY_INIT(VEXOpTelem, TYPE_USES_VEX_OPS);
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
};
}
File diff suppressed because it is too large. Load diff
+16 -40
View File
@@ -5,23 +5,18 @@ $end_info$
*/
#pragma once
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/Event.h>
#include <mutex>
#include <thread>
#include "Interface/Context/Context.h"
#include "Common/NetStream.h"
#include <FEXCore/Utils/Threads.h>
#include <atomic>
#include <istream>
#include <memory>
#include <mutex>
#include <stdint.h>
#include <string>
namespace FEXCore {
namespace Context {
struct Context;
}
class GdbServer {
public:
GdbServer(FEXCore::Context::Context *ctx);
@@ -29,24 +24,16 @@ public:
// Public for threading
void GdbServerLoop();
void AlertLibrariesChanged() {
LibraryMapChanged = true;
}
private:
void Break(int signal);
void OpenListenSocket();
std::unique_ptr<std::iostream> OpenSocket();
void StartThread();
std::string ReadPacket(std::iostream &stream);
void SendPacket(std::ostream &stream, const std::string& packet);
void SendPacket(std::ostream &stream, std::string packet);
void SendACK(std::ostream &stream, bool NACK);
Event ThreadBreakEvent{};
void WaitForThreadWakeup();
struct HandledPacketType {
std::string Response{};
enum ResponseType {
@@ -60,20 +47,18 @@ private:
ResponseType TypeResponse{};
};
void SendPacketPair(const HandledPacketType& packetPair);
HandledPacketType ProcessPacket(const std::string &packet);
HandledPacketType handleQuery(const std::string &packet);
HandledPacketType handleXfer(const std::string &packet);
HandledPacketType handleMemory(const std::string &packet);
HandledPacketType handleV(const std::string& packet);
HandledPacketType handleThreadOp(const std::string &packet);
HandledPacketType handleBreakpoint(const std::string &packet);
void SendPacketPair(HandledPacketType packetPair);
HandledPacketType ProcessPacket(std::string &packet);
HandledPacketType handleQuery(std::string &packet);
HandledPacketType handleXfer(std::string &packet);
HandledPacketType handleMemory(std::string &packet);
HandledPacketType handleV(std::string& packet);
HandledPacketType handleThreadOp(std::string &packet);
HandledPacketType handleBreakpoint(std::string &packet);
HandledPacketType handleProgramOffsets();
HandledPacketType ThreadAction(char action, uint32_t tid);
std::string readRegs();
HandledPacketType readReg(const std::string& packet);
HandledPacketType readReg(std::string& packet);
FEXCore::Context::Context *CTX;
std::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
@@ -81,17 +66,8 @@ private:
std::mutex sendMutex;
bool SettingNoAckMode{false};
bool NoAckMode{false};
bool NonStopMode{false};
std::string ThreadString{};
std::string OSDataString{};
void buildLibraryMap();
std::atomic<bool> LibraryMapChanged = true;
std::string LibraryMapString{};
// Used to keep track of which signals to pass to the guest
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
uint32_t CurrentDebuggingThread{};
int ListenSocket{};
FEX_CONFIG_OPT(Filename, APP_FILENAME);
};
+5 -139
View File
@@ -1,7 +1,6 @@
#include "Interface/Core/CPUID.h"
#include <FEXCore/Core/HostFeatures.h>
#include "Interface/Core/HostFeatures.h"
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
#ifdef _M_ARM_64
#include "aarch64/assembler-aarch64.h"
#include "aarch64/cpu-aarch64.h"
#include "aarch64/disasm-aarch64.h"
@@ -14,147 +13,14 @@
namespace FEXCore {
// Data Zero Prohibited flag
// 0b0 = ZVA/GVA/GZVA permitted
// 0b1 = ZVA/GVA/GZVA prohibited
constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
// Log2 of the blocksize in 32-bit words
constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
#ifdef _M_ARM_64
static uint32_t GetDCZID() {
uint64_t Result{};
__asm("mrs %[Res], DCZID_EL0"
: [Res] "=r" (Result));
return Result;
}
static uint32_t GetFPCR() {
uint64_t Result{};
__asm ("mrs %[Res], FPCR"
: [Res] "=r" (Result));
return Result;
}
static void SetFPCR(uint64_t Value) {
__asm ("msr FPCR, %[Value]"
:: [Value] "r" (Value));
}
#else
static uint32_t GetDCZID() {
// Return unsupported
return DCZID_DZP_MASK;
}
#endif
HostFeatures::HostFeatures() {
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
#ifdef VIXL_SIMULATOR
auto Features = vixl::CPUFeatures::All();
#else
auto Features = vixl::CPUFeatures::InferFromOS();
#endif
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
// Only supported when FEAT_AFP is supported
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
Supports3DNow = true;
SupportsSSE4A = true;
#ifdef VIXL_SIMULATOR
// Hardcode enable SVE with 256-bit wide registers.
SupportsAVX = true;
#else
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
#endif
SupportsSHA = true;
SupportsBMI1 = true;
SupportsBMI2 = true;
if (!SupportsAtomics) {
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
}
#ifdef _M_ARM_64
// We need to get the CPU's cache line size
// We expect sane targets that have correct cacheline sizes across clusters
uint64_t CTR;
__asm volatile ("mrs %[ctr], ctr_el0"
: [ctr] "=r"(CTR));
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
ICacheLineSize = 4 << (CTR & 0xF);
// Test if this CPU supports float exception trapping by attempting to enable
// On unsupported these bits are architecturally defined as RAZ/WI
constexpr uint32_t ExceptionEnableTraps =
(1U << 8) | // Invalid Operation float exception trap enable
(1U << 9) | // Divide by zero float exception trap enable
(1U << 10) | // Overflow float exception trap enable
(1U << 11) | // Underflow float exception trap enable
(1U << 12) | // Inexact float exception trap enable
(1U << 15); // Input Denormal float exception trap enable
uint32_t OriginalFPCR = GetFPCR();
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
SetFPCR(FPCR);
FPCR = GetFPCR();
SupportsFloatExceptions = (FPCR & ExceptionEnableTraps) == ExceptionEnableTraps;
// Set FPCR back to original just in case anything changed
SetFPCR(OriginalFPCR);
auto Features = vixl::CPUFeatures::InferFromOS();
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
#endif
#endif
#if defined(_M_X86_64) && !defined(VIXL_SIMULATOR)
#ifdef _M_X86_64
Xbyak::util::Cpu Features{};
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
SupportsRCPC = true;
SupportsTSOImm9 = true;
Supports3DNow = Features.has(Xbyak::util::Cpu::t3DN) && Features.has(Xbyak::util::Cpu::tE3DN);
SupportsSSE4A = Features.has(Xbyak::util::Cpu::tSSE4a);
SupportsAVX = true;
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
// xbyak doesn't know how to check for CLZero
uint32_t eax, ebx, ecx, edx;
// First ensure we support a new enough extended CPUID function range
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
if (eax >= 0x8000'0008U) {
// CLZero defined in 8000_00008_EBX[bit 0]
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
SupportsCLZERO = ebx & 1;
}
SupportsFlushInputsToZero = true;
SupportsFloatExceptions = true;
#endif
#ifdef VIXL_SIMULATOR
// simulator doesn't support dc(ZVA)
SupportsCLZERO = false;
#else
// Check if we can support cacheline clears
uint32_t DCZID = GetDCZID();
if ((DCZID & DCZID_DZP_MASK) == 0) {
uint32_t DCZID_Log2 = DCZID & DCZID_BS_MASK;
uint32_t DCZID_Bytes = (1 << DCZID_Log2) * sizeof(uint32_t);
// If the DC ZVA size matches the emulated cache line size
// This means we can use the instruction
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
}
#endif
}
}
+9
View File
@@ -0,0 +1,9 @@
#pragma once
namespace FEXCore {
class HostFeatures final {
public:
HostFeatures();
bool SupportsAES{};
};
}
File diff suppressed because it is too large. Load diff
@@ -1,777 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <FEXCore/Utils/BitUtils.h>
#include <cstdint>
namespace FEXCore::CPU {
#ifdef _M_X86_64
uint8_t AtomicFetchNeg(uint8_t *Addr) {
using Type = uint8_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
uint16_t AtomicFetchNeg(uint16_t *Addr) {
using Type = uint16_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
uint32_t AtomicFetchNeg(uint32_t *Addr) {
using Type = uint32_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
uint64_t AtomicFetchNeg(uint64_t *Addr) {
using Type = uint64_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
template<typename T>
T AtomicCompareAndSwap(T expected, T desired, T *addr)
{
std::atomic<T> *MemData = reinterpret_cast<std::atomic<T>*>(addr);
T Src1 = expected;
T Src2 = desired;
T Expected = Src1;
bool Result = MemData->compare_exchange_strong(Expected, Src2);
return Result ? Src1 : Expected;
}
template uint8_t AtomicCompareAndSwap<uint8_t>(uint8_t expected, uint8_t desired, uint8_t *addr);
template uint16_t AtomicCompareAndSwap<uint16_t>(uint16_t expected, uint16_t desired, uint16_t *addr);
template uint32_t AtomicCompareAndSwap<uint32_t>(uint32_t expected, uint32_t desired, uint32_t *addr);
template uint64_t AtomicCompareAndSwap<uint64_t>(uint64_t expected, uint64_t desired, uint64_t *addr);
#else
// Needs to match what the AArch64 JIT and unaligned signal handler expects
uint8_t AtomicFetchNeg(uint8_t *Addr) {
using Type = uint8_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxrb %w[Result], [%[Memory]];
neg %w[Tmp], %w[Result];
stlxrb %w[TmpStatus], %w[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
uint16_t AtomicFetchNeg(uint16_t *Addr) {
using Type = uint16_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxrh %w[Result], [%[Memory]];
neg %w[Tmp], %w[Result];
stlxrh %w[TmpStatus], %w[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
uint32_t AtomicFetchNeg(uint32_t *Addr) {
using Type = uint32_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxr %w[Result], [%[Memory]];
neg %w[Tmp], %w[Result];
stlxr %w[TmpStatus], %w[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
uint64_t AtomicFetchNeg(uint64_t *Addr) {
using Type = uint64_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxr %[Result], [%[Memory]];
neg %[Tmp], %[Result];
stlxr %w[TmpStatus], %[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
template<>
uint8_t AtomicCompareAndSwap(uint8_t expected, uint8_t desired, uint8_t *addr) {
using Type = uint8_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxrb %w[Tmp], [%[Memory]];
cmp %w[Tmp], %w[Expected], uxtb;
b.ne 2f;
stlxrb %w[Tmp2], %w[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %w[Result], %w[Expected];
b 3f;
2:
mov %w[Result], %w[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
template<>
uint16_t AtomicCompareAndSwap(uint16_t expected, uint16_t desired, uint16_t *addr) {
using Type = uint16_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxrh %w[Tmp], [%[Memory]];
cmp %w[Tmp], %w[Expected], uxth;
b.ne 2f;
stlxrh %w[Tmp2], %w[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %w[Result], %w[Expected];
b 3f;
2:
mov %w[Result], %w[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
template<>
uint32_t AtomicCompareAndSwap(uint32_t expected, uint32_t desired, uint32_t *addr) {
using Type = uint32_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxr %w[Tmp], [%[Memory]];
cmp %w[Tmp], %w[Expected];
b.ne 2f;
stlxr %w[Tmp2], %w[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %w[Result], %w[Expected];
b 3f;
2:
mov %w[Result], %w[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
template<>
uint64_t AtomicCompareAndSwap(uint64_t expected, uint64_t desired, uint64_t *addr) {
using Type = uint64_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxr %[Tmp], [%[Memory]];
cmp %[Tmp], %[Expected];
b.ne 2f;
stlxr %w[Tmp2], %[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %[Result], %[Expected];
b 3f;
2:
mov %[Result], %[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
#endif
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CASPair>();
// Size is the size of each pair element
switch (IROp->ElementSize) {
case 4: {
GD = AtomicCompareAndSwap(
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 8: {
std::atomic<__uint128_t> *MemData = *GetSrc<std::atomic<__uint128_t> **>(Data->SSAData, Op->Addr);
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Expected);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Desired);
__uint128_t Expected = Src1;
bool Result = MemData->compare_exchange_strong(Expected, Src2);
memcpy(GDP, Result ? &Src1 : &Expected, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", IROp->ElementSize); break;
}
}
DEF_OP(CAS) {
auto Op = IROp->C<IR::IROp_CAS>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
case 1: {
GD = AtomicCompareAndSwap(
*GetSrc<uint8_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint8_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint8_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 2: {
GD = AtomicCompareAndSwap(
*GetSrc<uint16_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint16_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint16_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 4: {
GD = AtomicCompareAndSwap(
*GetSrc<uint32_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint32_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint32_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 8: {
GD = AtomicCompareAndSwap(
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", OpSize); break;
}
}
DEF_OP(AtomicAdd) {
auto Op = IROp->C<IR::IROp_AtomicAdd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicSub) {
auto Op = IROp->C<IR::IROp_AtomicSub>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicAnd) {
auto Op = IROp->C<IR::IROp_AtomicAnd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicOr) {
auto Op = IROp->C<IR::IROp_AtomicOr>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicXor) {
auto Op = IROp->C<IR::IROp_AtomicXor>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicSwap) {
auto Op = IROp->C<IR::IROp_AtomicSwap>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchAdd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchSub) {
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchAnd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchXor) {
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchNeg) {
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
switch (IROp->Size) {
case 1: {
using Type = uint8_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
case 2: {
using Type = uint16_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
case 4: {
using Type = uint32_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
case 8: {
using Type = uint64_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,160 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include "Interface/HLE/Thunks/Thunks.h"
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <cstdint>
#include <unistd.h>
namespace FEXCore::CPU {
[[noreturn]]
static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
Thread->CTX->SignalThread(Thread, FEXCore::Core::SignalEvent::Return);
LOGMAN_MSG_A_FMT("unreachable");
FEX_UNREACHABLE;
}
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(SignalReturn) {
SignalReturn(Data->State);
}
DEF_OP(CallbackReturn) {
Data->State->CurrentFrame->Pointers.Interpreter.CallbackReturn(Data->State, Data->StackEntry);
}
DEF_OP(ExitFunction) {
auto Op = IROp->C<IR::IROp_ExitFunction>();
uint8_t OpSize = IROp->Size;
uintptr_t* ContextPtr = reinterpret_cast<uintptr_t*>(Data->State->CurrentFrame);
void *ContextData = reinterpret_cast<void*>(ContextPtr);
void *Src = GetSrc<void*>(Data->SSAData, Op->NewRIP);
memcpy(ContextData, Src, OpSize);
Data->BlockResults.Quit = true;
}
DEF_OP(Jump) {
auto Op = IROp->C<IR::IROp_Jump>();
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
const uintptr_t DataBegin = Data->CurrentIR->GetData();
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TargetBlock);
Data->BlockResults.Redo = true;
}
DEF_OP(CondJump) {
auto Op = IROp->C<IR::IROp_CondJump>();
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
const uintptr_t DataBegin = Data->CurrentIR->GetData();
bool CompResult;
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
if (Op->CompareSize == 4)
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
else
CompResult = IsConditionTrue<uint64_t, int64_t, double>(Op->Cond.Val, Src1, Src2);
if (CompResult) {
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TrueBlock);
}
else {
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->FalseBlock);
}
Data->BlockResults.Redo = true;
}
DEF_OP(Syscall) {
auto Op = IROp->C<IR::IROp_Syscall>();
FEXCore::HLE::SyscallArguments Args;
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
if (Op->Header.Args[j].IsInvalid()) break;
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
}
uint64_t Res = FEXCore::Context::HandleSyscall(Data->State->CTX->SyscallHandler, Data->State->CurrentFrame, &Args);
GD = Res;
}
DEF_OP(InlineSyscall) {
auto Op = IROp->C<IR::IROp_InlineSyscall>();
FEXCore::HLE::SyscallArguments Args;
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
if (Op->Header.Args[j].IsInvalid()) break;
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
}
// We don't want the errno handling but I also don't want to write inline ASM atm
uint64_t Res = syscall(
Op->HostSyscallNumber,
Args.Argument[0],
Args.Argument[1],
Args.Argument[2],
Args.Argument[3],
Args.Argument[4],
Args.Argument[5],
Args.Argument[6]
);
if (Res == -1) {
Res = -errno;
}
GD = Res;
}
DEF_OP(Thunk) {
auto Op = IROp->C<IR::IROp_Thunk>();
auto thunkFn = Data->State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
thunkFn(*GetSrc<void**>(Data->SSAData, Op->ArgPtr));
}
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
auto CodePtr = Data->CurrentEntry + Op->Offset;
if (memcmp((void*)CodePtr, &Op->CodeOriginalLow, Op->CodeLength) != 0) {
GD = 1;
} else {
GD = 0;
}
}
DEF_OP(ThreadRemoveCodeEntry) {
Data->State->CTX->ThreadRemoveCodeEntryFromJit(Data->State->CurrentFrame, Data->CurrentEntry);
}
DEF_OP(CPUID) {
auto Op = IROp->C<IR::IROp_CPUID>();
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
const uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Function);
const uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Leaf);
auto Results = Data->State->CTX->CPUID.RunFunction(Arg, Leaf);
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,261 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(VInsGPR) {
const auto Op = IROp->C<IR::IROp_VInsGPR>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto ElementSizeBits = ElementSize * 8;
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
const uint64_t Offset = Op->DestIdx * ElementSizeBits;
const auto InUpperLane = Offset >= SSEBitSize;
__uint128_t Mask = (1ULL << ElementSizeBits) - 1;
if (ElementSize == 8) {
Mask = ~0ULL;
}
const auto Src1 = *GetSrc<InterpVector256*>(Data->SSAData, Op->DestVector);
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
const auto Scalar = Src2 & Mask;
const auto ScaledOffset = InUpperLane ? Offset - SSEBitSize
: Offset;
// Now shift into place and set all bits but
// the ones where we're going to insert our value.
Mask <<= ScaledOffset;
Mask = ~Mask;
const auto Dst = [&] {
if (InUpperLane) {
return InterpVector256{
.Lower = Src1.Lower,
.Upper = (Src1.Upper & Mask) | (Scalar << ScaledOffset),
};
} else {
return InterpVector256{
.Lower = (Src1.Lower & Mask) | (Scalar << ScaledOffset),
.Upper = Src1.Upper,
};
}
}();
memcpy(GDP, &Dst, OpSize);
}
DEF_OP(VCastFromGPR) {
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Src), Op->Header.ElementSize);
}
DEF_OP(Float_FromGPR_S) {
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0404: { // Float <- int32_t
const float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
case 0x0408: { // Float <- int64_t
const float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
case 0x0804: { // Double <- int32_t
const double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
case 0x0808: { // Double <- int64_t
const double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
}
}
DEF_OP(Float_FToF) {
auto Op = IROp->C<IR::IROp_Float_FToF>();
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0804: { // Double <- Float
const double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Scalar);
memcpy(GDP, &Dst, 8);
break;
}
case 0x0408: { // Float <- Double
const float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Scalar);
memcpy(GDP, &Dst, 4);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
}
}
DEF_OP(Vector_SToF) {
auto Op = IROp->C<IR::IROp_Vector_SToF>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToZS) {
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToS) {
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToF) {
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint16_t ElementSize = Op->Header.ElementSize;
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (Conv) {
case 0x0804: { // Double <- float
// Only the lower elements from the source
// This uses half the source elements
uint8_t Elements = OpSize / 8;
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(double, float, Func, 0, 0)
break;
}
case 0x0408: { // Float <- Double
// Little bit tricky here
// Sometimes is used to convert from a 128bit vector register
// in to a 64bit vector register with different sized elements
// eg: %ssa5 i32v2 = Vector_FToF %ssa4 i128, #0x8
uint8_t Elements = (OpSize << 1) / Op->SrcElementSize;
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToI) {
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func_Nearest = [](auto a) { return std::rint(a); };
const auto Func_Neg = [](auto a) { return std::floor(a); };
const auto Func_Pos = [](auto a) { return std::ceil(a); };
const auto Func_Trunc = [](auto a) { return std::trunc(a); };
const auto Func_Host = [](auto a) { return std::rint(a); };
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
}
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
}
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
}
break;
case FEXCore::IR::Round_Towards_Zero.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
}
break;
case FEXCore::IR::Round_Host.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Host)
DO_VECTOR_1SRC_OP(8, double, Func_Host)
}
break;
}
memcpy(GDP, Tmp, OpSize);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,556 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace AES {
static __uint128_t InvShiftRows(uint8_t *State) {
uint8_t Shifted[16] = {
State[0], State[13], State[10], State[7],
State[4], State[1], State[14], State[11],
State[8], State[5], State[2], State[15],
State[12], State[9], State[6], State[3],
};
__uint128_t Res{};
memcpy(&Res, Shifted, 16);
return Res;
}
static __uint128_t InvSubBytes(uint8_t *State) {
// 16x16 matrix table
static const uint8_t InvSubstitutionTable[256] = {
0x52, 0x09, 0x6a, 0xd5, 0x30, 0x36, 0xa5, 0x38, 0xbf, 0x40, 0xa3, 0x9e, 0x81, 0xf3, 0xd7, 0xfb,
0x7c, 0xe3, 0x39, 0x82, 0x9b, 0x2f, 0xff, 0x87, 0x34, 0x8e, 0x43, 0x44, 0xc4, 0xde, 0xe9, 0xcb,
0x54, 0x7b, 0x94, 0x32, 0xa6, 0xc2, 0x23, 0x3d, 0xee, 0x4c, 0x95, 0x0b, 0x42, 0xfa, 0xc3, 0x4e,
0x08, 0x2e, 0xa1, 0x66, 0x28, 0xd9, 0x24, 0xb2, 0x76, 0x5b, 0xa2, 0x49, 0x6d, 0x8b, 0xd1, 0x25,
0x72, 0xf8, 0xf6, 0x64, 0x86, 0x68, 0x98, 0x16, 0xd4, 0xa4, 0x5c, 0xcc, 0x5d, 0x65, 0xb6, 0x92,
0x6c, 0x70, 0x48, 0x50, 0xfd, 0xed, 0xb9, 0xda, 0x5e, 0x15, 0x46, 0x57, 0xa7, 0x8d, 0x9d, 0x84,
0x90, 0xd8, 0xab, 0x00, 0x8c, 0xbc, 0xd3, 0x0a, 0xf7, 0xe4, 0x58, 0x05, 0xb8, 0xb3, 0x45, 0x06,
0xd0, 0x2c, 0x1e, 0x8f, 0xca, 0x3f, 0x0f, 0x02, 0xc1, 0xaf, 0xbd, 0x03, 0x01, 0x13, 0x8a, 0x6b,
0x3a, 0x91, 0x11, 0x41, 0x4f, 0x67, 0xdc, 0xea, 0x97, 0xf2, 0xcf, 0xce, 0xf0, 0xb4, 0xe6, 0x73,
0x96, 0xac, 0x74, 0x22, 0xe7, 0xad, 0x35, 0x85, 0xe2, 0xf9, 0x37, 0xe8, 0x1c, 0x75, 0xdf, 0x6e,
0x47, 0xf1, 0x1a, 0x71, 0x1d, 0x29, 0xc5, 0x89, 0x6f, 0xb7, 0x62, 0x0e, 0xaa, 0x18, 0xbe, 0x1b,
0xfc, 0x56, 0x3e, 0x4b, 0xc6, 0xd2, 0x79, 0x20, 0x9a, 0xdb, 0xc0, 0xfe, 0x78, 0xcd, 0x5a, 0xf4,
0x1f, 0xdd, 0xa8, 0x33, 0x88, 0x07, 0xc7, 0x31, 0xb1, 0x12, 0x10, 0x59, 0x27, 0x80, 0xec, 0x5f,
0x60, 0x51, 0x7f, 0xa9, 0x19, 0xb5, 0x4a, 0x0d, 0x2d, 0xe5, 0x7a, 0x9f, 0x93, 0xc9, 0x9c, 0xef,
0xa0, 0xe0, 0x3b, 0x4d, 0xae, 0x2a, 0xf5, 0xb0, 0xc8, 0xeb, 0xbb, 0x3c, 0x83, 0x53, 0x99, 0x61,
0x17, 0x2b, 0x04, 0x7e, 0xba, 0x77, 0xd6, 0x26, 0xe1, 0x69, 0x14, 0x63, 0x55, 0x21, 0x0c, 0x7d,
};
// Uses a byte substitution table with a constant set of values
// Needs to do a table look up
uint8_t Substituted[16];
for (size_t i = 0; i < 16; ++i) {
Substituted[i] = InvSubstitutionTable[State[i]];
}
__uint128_t Res{};
memcpy(&Res, Substituted, 16);
return Res;
}
static __uint128_t ShiftRows(uint8_t *State) {
uint8_t Shifted[16] = {
State[0], State[5], State[10], State[15],
State[4], State[9], State[14], State[3],
State[8], State[13], State[2], State[7],
State[12], State[1], State[6], State[11],
};
__uint128_t Res{};
memcpy(&Res, Shifted, 16);
return Res;
}
static __uint128_t SubBytes(uint8_t *State, size_t Bytes) {
// 16x16 matrix table
static const uint8_t SubstitutionTable[256] = {
0x63, 0x7c, 0x77, 0x7b, 0xf2, 0x6b, 0x6f, 0xc5, 0x30, 0x01, 0x67, 0x2b, 0xfe, 0xd7, 0xab, 0x76,
0xca, 0x82, 0xc9, 0x7d, 0xfa, 0x59, 0x47, 0xf0, 0xad, 0xd4, 0xa2, 0xaf, 0x9c, 0xa4, 0x72, 0xc0,
0xb7, 0xfd, 0x93, 0x26, 0x36, 0x3f, 0xf7, 0xcc, 0x34, 0xa5, 0xe5, 0xf1, 0x71, 0xd8, 0x31, 0x15,
0x04, 0xc7, 0x23, 0xc3, 0x18, 0x96, 0x05, 0x9a, 0x07, 0x12, 0x80, 0xe2, 0xeb, 0x27, 0xb2, 0x75,
0x09, 0x83, 0x2c, 0x1a, 0x1b, 0x6e, 0x5a, 0xa0, 0x52, 0x3b, 0xd6, 0xb3, 0x29, 0xe3, 0x2f, 0x84,
0x53, 0xd1, 0x00, 0xed, 0x20, 0xfc, 0xb1, 0x5b, 0x6a, 0xcb, 0xbe, 0x39, 0x4a, 0x4c, 0x58, 0xcf,
0xd0, 0xef, 0xaa, 0xfb, 0x43, 0x4d, 0x33, 0x85, 0x45, 0xf9, 0x02, 0x7f, 0x50, 0x3c, 0x9f, 0xa8,
0x51, 0xa3, 0x40, 0x8f, 0x92, 0x9d, 0x38, 0xf5, 0xbc, 0xb6, 0xda, 0x21, 0x10, 0xff, 0xf3, 0xd2,
0xcd, 0x0c, 0x13, 0xec, 0x5f, 0x97, 0x44, 0x17, 0xc4, 0xa7, 0x7e, 0x3d, 0x64, 0x5d, 0x19, 0x73,
0x60, 0x81, 0x4f, 0xdc, 0x22, 0x2a, 0x90, 0x88, 0x46, 0xee, 0xb8, 0x14, 0xde, 0x5e, 0x0b, 0xdb,
0xe0, 0x32, 0x3a, 0x0a, 0x49, 0x06, 0x24, 0x5c, 0xc2, 0xd3, 0xac, 0x62, 0x91, 0x95, 0xe4, 0x79,
0xe7, 0xc8, 0x37, 0x6d, 0x8d, 0xd5, 0x4e, 0xa9, 0x6c, 0x56, 0xf4, 0xea, 0x65, 0x7a, 0xae, 0x08,
0xba, 0x78, 0x25, 0x2e, 0x1c, 0xa6, 0xb4, 0xc6, 0xe8, 0xdd, 0x74, 0x1f, 0x4b, 0xbd, 0x8b, 0x8a,
0x70, 0x3e, 0xb5, 0x66, 0x48, 0x03, 0xf6, 0x0e, 0x61, 0x35, 0x57, 0xb9, 0x86, 0xc1, 0x1d, 0x9e,
0xe1, 0xf8, 0x98, 0x11, 0x69, 0xd9, 0x8e, 0x94, 0x9b, 0x1e, 0x87, 0xe9, 0xce, 0x55, 0x28, 0xdf,
0x8c, 0xa1, 0x89, 0x0d, 0xbf, 0xe6, 0x42, 0x68, 0x41, 0x99, 0x2d, 0x0f, 0xb0, 0x54, 0xbb, 0x16,
};
// Uses a byte substitution table with a constant set of values
// Needs to do a table look up
uint8_t Substituted[16];
Bytes = std::min(Bytes, (size_t)16);
for (size_t i = 0; i < Bytes; ++i) {
Substituted[i] = SubstitutionTable[State[i]];
}
__uint128_t Res{};
memcpy(&Res, Substituted, Bytes);
return Res;
}
static uint8_t FFMul02(uint8_t in) {
static const uint8_t FFMul02[256] = {
0x00, 0x02, 0x04, 0x06, 0x08, 0x0a, 0x0c, 0x0e, 0x10, 0x12, 0x14, 0x16, 0x18, 0x1a, 0x1c, 0x1e,
0x20, 0x22, 0x24, 0x26, 0x28, 0x2a, 0x2c, 0x2e, 0x30, 0x32, 0x34, 0x36, 0x38, 0x3a, 0x3c, 0x3e,
0x40, 0x42, 0x44, 0x46, 0x48, 0x4a, 0x4c, 0x4e, 0x50, 0x52, 0x54, 0x56, 0x58, 0x5a, 0x5c, 0x5e,
0x60, 0x62, 0x64, 0x66, 0x68, 0x6a, 0x6c, 0x6e, 0x70, 0x72, 0x74, 0x76, 0x78, 0x7a, 0x7c, 0x7e,
0x80, 0x82, 0x84, 0x86, 0x88, 0x8a, 0x8c, 0x8e, 0x90, 0x92, 0x94, 0x96, 0x98, 0x9a, 0x9c, 0x9e,
0xa0, 0xa2, 0xa4, 0xa6, 0xa8, 0xaa, 0xac, 0xae, 0xb0, 0xb2, 0xb4, 0xb6, 0xb8, 0xba, 0xbc, 0xbe,
0xc0, 0xc2, 0xc4, 0xc6, 0xc8, 0xca, 0xcc, 0xce, 0xd0, 0xd2, 0xd4, 0xd6, 0xd8, 0xda, 0xdc, 0xde,
0xe0, 0xe2, 0xe4, 0xe6, 0xe8, 0xea, 0xec, 0xee, 0xf0, 0xf2, 0xf4, 0xf6, 0xf8, 0xfa, 0xfc, 0xfe,
0x1b, 0x19, 0x1f, 0x1d, 0x13, 0x11, 0x17, 0x15, 0x0b, 0x09, 0x0f, 0x0d, 0x03, 0x01, 0x07, 0x05,
0x3b, 0x39, 0x3f, 0x3d, 0x33, 0x31, 0x37, 0x35, 0x2b, 0x29, 0x2f, 0x2d, 0x23, 0x21, 0x27, 0x25,
0x5b, 0x59, 0x5f, 0x5d, 0x53, 0x51, 0x57, 0x55, 0x4b, 0x49, 0x4f, 0x4d, 0x43, 0x41, 0x47, 0x45,
0x7b, 0x79, 0x7f, 0x7d, 0x73, 0x71, 0x77, 0x75, 0x6b, 0x69, 0x6f, 0x6d, 0x63, 0x61, 0x67, 0x65,
0x9b, 0x99, 0x9f, 0x9d, 0x93, 0x91, 0x97, 0x95, 0x8b, 0x89, 0x8f, 0x8d, 0x83, 0x81, 0x87, 0x85,
0xbb, 0xb9, 0xbf, 0xbd, 0xb3, 0xb1, 0xb7, 0xb5, 0xab, 0xa9, 0xaf, 0xad, 0xa3, 0xa1, 0xa7, 0xa5,
0xdb, 0xd9, 0xdf, 0xdd, 0xd3, 0xd1, 0xd7, 0xd5, 0xcb, 0xc9, 0xcf, 0xcd, 0xc3, 0xc1, 0xc7, 0xc5,
0xfb, 0xf9, 0xff, 0xfd, 0xf3, 0xf1, 0xf7, 0xf5, 0xeb, 0xe9, 0xef, 0xed, 0xe3, 0xe1, 0xe7, 0xe5,
};
return FFMul02[in];
}
static uint8_t FFMul03(uint8_t in) {
static const uint8_t FFMul03[256] = {
0x00, 0x03, 0x06, 0x05, 0x0c, 0x0f, 0x0a, 0x09, 0x18, 0x1b, 0x1e, 0x1d, 0x14, 0x17, 0x12, 0x11,
0x30, 0x33, 0x36, 0x35, 0x3c, 0x3f, 0x3a, 0x39, 0x28, 0x2b, 0x2e, 0x2d, 0x24, 0x27, 0x22, 0x21,
0x60, 0x63, 0x66, 0x65, 0x6c, 0x6f, 0x6a, 0x69, 0x78, 0x7b, 0x7e, 0x7d, 0x74, 0x77, 0x72, 0x71,
0x50, 0x53, 0x56, 0x55, 0x5c, 0x5f, 0x5a, 0x59, 0x48, 0x4b, 0x4e, 0x4d, 0x44, 0x47, 0x42, 0x41,
0xc0, 0xc3, 0xc6, 0xc5, 0xcc, 0xcf, 0xca, 0xc9, 0xd8, 0xdb, 0xde, 0xdd, 0xd4, 0xd7, 0xd2, 0xd1,
0xf0, 0xf3, 0xf6, 0xf5, 0xfc, 0xff, 0xfa, 0xf9, 0xe8, 0xeb, 0xee, 0xed, 0xe4, 0xe7, 0xe2, 0xe1,
0xa0, 0xa3, 0xa6, 0xa5, 0xac, 0xaf, 0xaa, 0xa9, 0xb8, 0xbb, 0xbe, 0xbd, 0xb4, 0xb7, 0xb2, 0xb1,
0x90, 0x93, 0x96, 0x95, 0x9c, 0x9f, 0x9a, 0x99, 0x88, 0x8b, 0x8e, 0x8d, 0x84, 0x87, 0x82, 0x81,
0x9b, 0x98, 0x9d, 0x9e, 0x97, 0x94, 0x91, 0x92, 0x83, 0x80, 0x85, 0x86, 0x8f, 0x8c, 0x89, 0x8a,
0xab, 0xa8, 0xad, 0xae, 0xa7, 0xa4, 0xa1, 0xa2, 0xb3, 0xb0, 0xb5, 0xb6, 0xbf, 0xbc, 0xb9, 0xba,
0xfb, 0xf8, 0xfd, 0xfe, 0xf7, 0xf4, 0xf1, 0xf2, 0xe3, 0xe0, 0xe5, 0xe6, 0xef, 0xec, 0xe9, 0xea,
0xcb, 0xc8, 0xcd, 0xce, 0xc7, 0xc4, 0xc1, 0xc2, 0xd3, 0xd0, 0xd5, 0xd6, 0xdf, 0xdc, 0xd9, 0xda,
0x5b, 0x58, 0x5d, 0x5e, 0x57, 0x54, 0x51, 0x52, 0x43, 0x40, 0x45, 0x46, 0x4f, 0x4c, 0x49, 0x4a,
0x6b, 0x68, 0x6d, 0x6e, 0x67, 0x64, 0x61, 0x62, 0x73, 0x70, 0x75, 0x76, 0x7f, 0x7c, 0x79, 0x7a,
0x3b, 0x38, 0x3d, 0x3e, 0x37, 0x34, 0x31, 0x32, 0x23, 0x20, 0x25, 0x26, 0x2f, 0x2c, 0x29, 0x2a,
0x0b, 0x08, 0x0d, 0x0e, 0x07, 0x04, 0x01, 0x02, 0x13, 0x10, 0x15, 0x16, 0x1f, 0x1c, 0x19, 0x1a,
};
return FFMul03[in];
}
static __uint128_t MixColumns(uint8_t *State) {
uint8_t In0[16] = {
State[0], State[4], State[8], State[12],
State[1], State[5], State[9], State[13],
State[2], State[6], State[10], State[14],
State[3], State[7], State[11], State[15],
};
uint8_t Out0[4]{};
uint8_t Out1[4]{};
uint8_t Out2[4]{};
uint8_t Out3[4]{};
for (size_t i = 0; i < 4; ++i) {
Out0[i] = FFMul02(In0[0 + i]) ^ FFMul03(In0[4 + i]) ^ In0[8 + i] ^ In0[12 + i];
Out1[i] = In0[0 + i] ^ FFMul02(In0[4 + i]) ^ FFMul03(In0[8 + i]) ^ In0[12 + i];
Out2[i] = In0[0 + i] ^ In0[4 + i] ^ FFMul02(In0[8 + i]) ^ FFMul03(In0[12 + i]);
Out3[i] = FFMul03(In0[0 + i]) ^ In0[4 + i] ^ In0[8 + i] ^ FFMul02(In0[12 + i]);
}
uint8_t OutArray[16] = {
Out0[0], Out1[0], Out2[0], Out3[0],
Out0[1], Out1[1], Out2[1], Out3[1],
Out0[2], Out1[2], Out2[2], Out3[2],
Out0[3], Out1[3], Out2[3], Out3[3],
};
__uint128_t Res{};
memcpy(&Res, OutArray, 16);
return Res;
}
static uint8_t FFMul09(uint8_t in) {
static const uint8_t FFMul09[256] = {
0x00, 0x09, 0x12, 0x1b, 0x24, 0x2d, 0x36, 0x3f, 0x48, 0x41, 0x5a, 0x53, 0x6c, 0x65, 0x7e, 0x77,
0x90, 0x99, 0x82, 0x8b, 0xb4, 0xbd, 0xa6, 0xaf, 0xd8, 0xd1, 0xca, 0xc3, 0xfc, 0xf5, 0xee, 0xe7,
0x3b, 0x32, 0x29, 0x20, 0x1f, 0x16, 0x0d, 0x04, 0x73, 0x7a, 0x61, 0x68, 0x57, 0x5e, 0x45, 0x4c,
0xab, 0xa2, 0xb9, 0xb0, 0x8f, 0x86, 0x9d, 0x94, 0xe3, 0xea, 0xf1, 0xf8, 0xc7, 0xce, 0xd5, 0xdc,
0x76, 0x7f, 0x64, 0x6d, 0x52, 0x5b, 0x40, 0x49, 0x3e, 0x37, 0x2c, 0x25, 0x1a, 0x13, 0x08, 0x01,
0xe6, 0xef, 0xf4, 0xfd, 0xc2, 0xcb, 0xd0, 0xd9, 0xae, 0xa7, 0xbc, 0xb5, 0x8a, 0x83, 0x98, 0x91,
0x4d, 0x44, 0x5f, 0x56, 0x69, 0x60, 0x7b, 0x72, 0x05, 0x0c, 0x17, 0x1e, 0x21, 0x28, 0x33, 0x3a,
0xdd, 0xd4, 0xcf, 0xc6, 0xf9, 0xf0, 0xeb, 0xe2, 0x95, 0x9c, 0x87, 0x8e, 0xb1, 0xb8, 0xa3, 0xaa,
0xec, 0xe5, 0xfe, 0xf7, 0xc8, 0xc1, 0xda, 0xd3, 0xa4, 0xad, 0xb6, 0xbf, 0x80, 0x89, 0x92, 0x9b,
0x7c, 0x75, 0x6e, 0x67, 0x58, 0x51, 0x4a, 0x43, 0x34, 0x3d, 0x26, 0x2f, 0x10, 0x19, 0x02, 0x0b,
0xd7, 0xde, 0xc5, 0xcc, 0xf3, 0xfa, 0xe1, 0xe8, 0x9f, 0x96, 0x8d, 0x84, 0xbb, 0xb2, 0xa9, 0xa0,
0x47, 0x4e, 0x55, 0x5c, 0x63, 0x6a, 0x71, 0x78, 0x0f, 0x06, 0x1d, 0x14, 0x2b, 0x22, 0x39, 0x30,
0x9a, 0x93, 0x88, 0x81, 0xbe, 0xb7, 0xac, 0xa5, 0xd2, 0xdb, 0xc0, 0xc9, 0xf6, 0xff, 0xe4, 0xed,
0x0a, 0x03, 0x18, 0x11, 0x2e, 0x27, 0x3c, 0x35, 0x42, 0x4b, 0x50, 0x59, 0x66, 0x6f, 0x74, 0x7d,
0xa1, 0xa8, 0xb3, 0xba, 0x85, 0x8c, 0x97, 0x9e, 0xe9, 0xe0, 0xfb, 0xf2, 0xcd, 0xc4, 0xdf, 0xd6,
0x31, 0x38, 0x23, 0x2a, 0x15, 0x1c, 0x07, 0x0e, 0x79, 0x70, 0x6b, 0x62, 0x5d, 0x54, 0x4f, 0x46,
};
return FFMul09[in];
}
static uint8_t FFMul0B(uint8_t in) {
static const uint8_t FFMul0B[256] = {
0x00, 0x0b, 0x16, 0x1d, 0x2c, 0x27, 0x3a, 0x31, 0x58, 0x53, 0x4e, 0x45, 0x74, 0x7f, 0x62, 0x69,
0xb0, 0xbb, 0xa6, 0xad, 0x9c, 0x97, 0x8a, 0x81, 0xe8, 0xe3, 0xfe, 0xf5, 0xc4, 0xcf, 0xd2, 0xd9,
0x7b, 0x70, 0x6d, 0x66, 0x57, 0x5c, 0x41, 0x4a, 0x23, 0x28, 0x35, 0x3e, 0x0f, 0x04, 0x19, 0x12,
0xcb, 0xc0, 0xdd, 0xd6, 0xe7, 0xec, 0xf1, 0xfa, 0x93, 0x98, 0x85, 0x8e, 0xbf, 0xb4, 0xa9, 0xa2,
0xf6, 0xfd, 0xe0, 0xeb, 0xda, 0xd1, 0xcc, 0xc7, 0xae, 0xa5, 0xb8, 0xb3, 0x82, 0x89, 0x94, 0x9f,
0x46, 0x4d, 0x50, 0x5b, 0x6a, 0x61, 0x7c, 0x77, 0x1e, 0x15, 0x08, 0x03, 0x32, 0x39, 0x24, 0x2f,
0x8d, 0x86, 0x9b, 0x90, 0xa1, 0xaa, 0xb7, 0xbc, 0xd5, 0xde, 0xc3, 0xc8, 0xf9, 0xf2, 0xef, 0xe4,
0x3d, 0x36, 0x2b, 0x20, 0x11, 0x1a, 0x07, 0x0c, 0x65, 0x6e, 0x73, 0x78, 0x49, 0x42, 0x5f, 0x54,
0xf7, 0xfc, 0xe1, 0xea, 0xdb, 0xd0, 0xcd, 0xc6, 0xaf, 0xa4, 0xb9, 0xb2, 0x83, 0x88, 0x95, 0x9e,
0x47, 0x4c, 0x51, 0x5a, 0x6b, 0x60, 0x7d, 0x76, 0x1f, 0x14, 0x09, 0x02, 0x33, 0x38, 0x25, 0x2e,
0x8c, 0x87, 0x9a, 0x91, 0xa0, 0xab, 0xb6, 0xbd, 0xd4, 0xdf, 0xc2, 0xc9, 0xf8, 0xf3, 0xee, 0xe5,
0x3c, 0x37, 0x2a, 0x21, 0x10, 0x1b, 0x06, 0x0d, 0x64, 0x6f, 0x72, 0x79, 0x48, 0x43, 0x5e, 0x55,
0x01, 0x0a, 0x17, 0x1c, 0x2d, 0x26, 0x3b, 0x30, 0x59, 0x52, 0x4f, 0x44, 0x75, 0x7e, 0x63, 0x68,
0xb1, 0xba, 0xa7, 0xac, 0x9d, 0x96, 0x8b, 0x80, 0xe9, 0xe2, 0xff, 0xf4, 0xc5, 0xce, 0xd3, 0xd8,
0x7a, 0x71, 0x6c, 0x67, 0x56, 0x5d, 0x40, 0x4b, 0x22, 0x29, 0x34, 0x3f, 0x0e, 0x05, 0x18, 0x13,
0xca, 0xc1, 0xdc, 0xd7, 0xe6, 0xed, 0xf0, 0xfb, 0x92, 0x99, 0x84, 0x8f, 0xbe, 0xb5, 0xa8, 0xa3,
};
return FFMul0B[in];
}
static uint8_t FFMul0D(uint8_t in) {
static const uint8_t FFMul0D[256] = {
0x00, 0x0d, 0x1a, 0x17, 0x34, 0x39, 0x2e, 0x23, 0x68, 0x65, 0x72, 0x7f, 0x5c, 0x51, 0x46, 0x4b,
0xd0, 0xdd, 0xca, 0xc7, 0xe4, 0xe9, 0xfe, 0xf3, 0xb8, 0xb5, 0xa2, 0xaf, 0x8c, 0x81, 0x96, 0x9b,
0xbb, 0xb6, 0xa1, 0xac, 0x8f, 0x82, 0x95, 0x98, 0xd3, 0xde, 0xc9, 0xc4, 0xe7, 0xea, 0xfd, 0xf0,
0x6b, 0x66, 0x71, 0x7c, 0x5f, 0x52, 0x45, 0x48, 0x03, 0x0e, 0x19, 0x14, 0x37, 0x3a, 0x2d, 0x20,
0x6d, 0x60, 0x77, 0x7a, 0x59, 0x54, 0x43, 0x4e, 0x05, 0x08, 0x1f, 0x12, 0x31, 0x3c, 0x2b, 0x26,
0xbd, 0xb0, 0xa7, 0xaa, 0x89, 0x84, 0x93, 0x9e, 0xd5, 0xd8, 0xcf, 0xc2, 0xe1, 0xec, 0xfb, 0xf6,
0xd6, 0xdb, 0xcc, 0xc1, 0xe2, 0xef, 0xf8, 0xf5, 0xbe, 0xb3, 0xa4, 0xa9, 0x8a, 0x87, 0x90, 0x9d,
0x06, 0x0b, 0x1c, 0x11, 0x32, 0x3f, 0x28, 0x25, 0x6e, 0x63, 0x74, 0x79, 0x5a, 0x57, 0x40, 0x4d,
0xda, 0xd7, 0xc0, 0xcd, 0xee, 0xe3, 0xf4, 0xf9, 0xb2, 0xbf, 0xa8, 0xa5, 0x86, 0x8b, 0x9c, 0x91,
0x0a, 0x07, 0x10, 0x1d, 0x3e, 0x33, 0x24, 0x29, 0x62, 0x6f, 0x78, 0x75, 0x56, 0x5b, 0x4c, 0x41,
0x61, 0x6c, 0x7b, 0x76, 0x55, 0x58, 0x4f, 0x42, 0x09, 0x04, 0x13, 0x1e, 0x3d, 0x30, 0x27, 0x2a,
0xb1, 0xbc, 0xab, 0xa6, 0x85, 0x88, 0x9f, 0x92, 0xd9, 0xd4, 0xc3, 0xce, 0xed, 0xe0, 0xf7, 0xfa,
0xb7, 0xba, 0xad, 0xa0, 0x83, 0x8e, 0x99, 0x94, 0xdf, 0xd2, 0xc5, 0xc8, 0xeb, 0xe6, 0xf1, 0xfc,
0x67, 0x6a, 0x7d, 0x70, 0x53, 0x5e, 0x49, 0x44, 0x0f, 0x02, 0x15, 0x18, 0x3b, 0x36, 0x21, 0x2c,
0x0c, 0x01, 0x16, 0x1b, 0x38, 0x35, 0x22, 0x2f, 0x64, 0x69, 0x7e, 0x73, 0x50, 0x5d, 0x4a, 0x47,
0xdc, 0xd1, 0xc6, 0xcb, 0xe8, 0xe5, 0xf2, 0xff, 0xb4, 0xb9, 0xae, 0xa3, 0x80, 0x8d, 0x9a, 0x97,
};
return FFMul0D[in];
}
static uint8_t FFMul0E(uint8_t in) {
static const uint8_t FFMul0E[256] = {
0x00, 0x0e, 0x1c, 0x12, 0x38, 0x36, 0x24, 0x2a, 0x70, 0x7e, 0x6c, 0x62, 0x48, 0x46, 0x54, 0x5a,
0xe0, 0xee, 0xfc, 0xf2, 0xd8, 0xd6, 0xc4, 0xca, 0x90, 0x9e, 0x8c, 0x82, 0xa8, 0xa6, 0xb4, 0xba,
0xdb, 0xd5, 0xc7, 0xc9, 0xe3, 0xed, 0xff, 0xf1, 0xab, 0xa5, 0xb7, 0xb9, 0x93, 0x9d, 0x8f, 0x81,
0x3b, 0x35, 0x27, 0x29, 0x03, 0x0d, 0x1f, 0x11, 0x4b, 0x45, 0x57, 0x59, 0x73, 0x7d, 0x6f, 0x61,
0xad, 0xa3, 0xb1, 0xbf, 0x95, 0x9b, 0x89, 0x87, 0xdd, 0xd3, 0xc1, 0xcf, 0xe5, 0xeb, 0xf9, 0xf7,
0x4d, 0x43, 0x51, 0x5f, 0x75, 0x7b, 0x69, 0x67, 0x3d, 0x33, 0x21, 0x2f, 0x05, 0x0b, 0x19, 0x17,
0x76, 0x78, 0x6a, 0x64, 0x4e, 0x40, 0x52, 0x5c, 0x06, 0x08, 0x1a, 0x14, 0x3e, 0x30, 0x22, 0x2c,
0x96, 0x98, 0x8a, 0x84, 0xae, 0xa0, 0xb2, 0xbc, 0xe6, 0xe8, 0xfa, 0xf4, 0xde, 0xd0, 0xc2, 0xcc,
0x41, 0x4f, 0x5d, 0x53, 0x79, 0x77, 0x65, 0x6b, 0x31, 0x3f, 0x2d, 0x23, 0x09, 0x07, 0x15, 0x1b,
0xa1, 0xaf, 0xbd, 0xb3, 0x99, 0x97, 0x85, 0x8b, 0xd1, 0xdf, 0xcd, 0xc3, 0xe9, 0xe7, 0xf5, 0xfb,
0x9a, 0x94, 0x86, 0x88, 0xa2, 0xac, 0xbe, 0xb0, 0xea, 0xe4, 0xf6, 0xf8, 0xd2, 0xdc, 0xce, 0xc0,
0x7a, 0x74, 0x66, 0x68, 0x42, 0x4c, 0x5e, 0x50, 0x0a, 0x04, 0x16, 0x18, 0x32, 0x3c, 0x2e, 0x20,
0xec, 0xe2, 0xf0, 0xfe, 0xd4, 0xda, 0xc8, 0xc6, 0x9c, 0x92, 0x80, 0x8e, 0xa4, 0xaa, 0xb8, 0xb6,
0x0c, 0x02, 0x10, 0x1e, 0x34, 0x3a, 0x28, 0x26, 0x7c, 0x72, 0x60, 0x6e, 0x44, 0x4a, 0x58, 0x56,
0x37, 0x39, 0x2b, 0x25, 0x0f, 0x01, 0x13, 0x1d, 0x47, 0x49, 0x5b, 0x55, 0x7f, 0x71, 0x63, 0x6d,
0xd7, 0xd9, 0xcb, 0xc5, 0xef, 0xe1, 0xf3, 0xfd, 0xa7, 0xa9, 0xbb, 0xb5, 0x9f, 0x91, 0x83, 0x8d,
};
return FFMul0E[in];
}
static __uint128_t InvMixColumns(uint8_t *State) {
uint8_t In0[16] = {
State[0], State[4], State[8], State[12],
State[1], State[5], State[9], State[13],
State[2], State[6], State[10], State[14],
State[3], State[7], State[11], State[15],
};
uint8_t Out0[4]{};
uint8_t Out1[4]{};
uint8_t Out2[4]{};
uint8_t Out3[4]{};
for (size_t i = 0; i < 4; ++i) {
Out0[i] = FFMul0E(In0[0 + i]) ^ FFMul0B(In0[4 + i]) ^ FFMul0D(In0[8 + i]) ^ FFMul09(In0[12 + i]);
Out1[i] = FFMul09(In0[0 + i]) ^ FFMul0E(In0[4 + i]) ^ FFMul0B(In0[8 + i]) ^ FFMul0D(In0[12 + i]);
Out2[i] = FFMul0D(In0[0 + i]) ^ FFMul09(In0[4 + i]) ^ FFMul0E(In0[8 + i]) ^ FFMul0B(In0[12 + i]);
Out3[i] = FFMul0B(In0[0 + i]) ^ FFMul0D(In0[4 + i]) ^ FFMul09(In0[8 + i]) ^ FFMul0E(In0[12 + i]);
}
uint8_t OutArray[16] = {
Out0[0], Out1[0], Out2[0], Out3[0],
Out0[1], Out1[1], Out2[1], Out3[1],
Out0[2], Out1[2], Out2[2], Out3[2],
Out0[3], Out1[3], Out2[3], Out3[3],
};
__uint128_t Res{};
memcpy(&Res, OutArray, 16);
return Res;
}
}
namespace CRC32 {
// CRC32 per byte lookup table.
constexpr std::array<uint32_t, 256> CRC32CTable = []() consteval {
std::array<uint32_t, 256> Table{};
// Clang 11.x doesn't support bitreverse as a consteval
// constexpr uint32_t Polynomial = 0x1EDC6F41;
constexpr uint32_t PolynomialRev = 0x82F63B78; //__builtin_bitreverse32(Polynomial);
for (size_t Char = 0; Char < std::size(Table); ++Char) {
uint32_t CurrentChar = Char;
for (size_t i = 0; i < 8; ++i) {
if (CurrentChar & 1) {
CurrentChar = (CurrentChar >> 1) ^ PolynomialRev;
}
else {
CurrentChar >>= 1;
}
}
Table[Char] = CurrentChar;
}
return Table;
}();
uint32_t crc32cb(uint32_t Accumulator, uint8_t data) {
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ data] ^ Accumulator >> 8;
return Accumulator;
}
uint32_t crc32ch(uint32_t Accumulator, uint16_t data) {
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
return Accumulator;
}
uint32_t crc32cw(uint32_t Accumulator, uint32_t data) {
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
return Accumulator;
}
uint32_t crc32cx(uint32_t Accumulator, uint64_t data) {
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 32) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 40) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 48) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 56) & 0xFF)] ^ Accumulator >> 8;
return Accumulator;
}
}
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(AESImc) {
auto Op = IROp->C<IR::IROp_VAESImc>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
// Pseudo-code
// Dst = InvMixColumns(STATE)
__uint128_t Tmp{};
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Src1));
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESEnc) {
auto Op = IROp->C<IR::IROp_VAESEnc>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = ShiftRows(STATE)
// STATE = SubBytes(STATE)
// STATE = MixColumns(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
Tmp = AES::MixColumns(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESEncLast) {
auto Op = IROp->C<IR::IROp_VAESEncLast>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = ShiftRows(STATE)
// STATE = SubBytes(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESDec) {
auto Op = IROp->C<IR::IROp_VAESDec>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = InvShiftRows(STATE)
// STATE = InvSubBytes(STATE)
// STATE = InvMixColumns(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESDecLast) {
auto Op = IROp->C<IR::IROp_VAESDecLast>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = InvShiftRows(STATE)
// STATE = InvSubBytes(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESKeyGenAssist) {
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Src);
// Pseudo-code
// X3 = Src1[127:96]
// X2 = Src1[95:64]
// X1 = Src1[63:32]
// X0 = Src1[31:30]
// RCON = (Zext)rcon
// Dest[31:0] = SubWord(X1)
// Dest[63:32] = RotWord(SubWord(X1)) XOR RCON
// Dest[95:64] = SubWord(X3)
// Dest[127:96] = RotWord(SubWord(X3)) XOR RCON
__uint128_t Tmp{};
uint32_t X1{};
uint32_t X3{};
memcpy(&X1, &Src1[4], 4);
memcpy(&X3, &Src1[12], 4);
uint32_t SubWord_X1 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X1), 4);
uint32_t SubWord_X3 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X3), 4);
auto Ror = [] (auto In, auto R) {
auto RotateMask = sizeof(In) * 8 - 1;
R &= RotateMask;
return (In >> R) | (In << (sizeof(In) * 8 - R));
};
uint32_t Rot_X1 = Ror(SubWord_X1, 8);
uint32_t Rot_X3 = Ror(SubWord_X3, 8);
Tmp = Rot_X3 ^ Op->RCON;
Tmp <<= 32;
Tmp |= SubWord_X3;
Tmp <<= 32;
Tmp |= Rot_X1 ^ Op->RCON;
Tmp <<= 32;
Tmp |= SubWord_X1;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(CRC32) {
auto Op = IROp->C<IR::IROp_CRC32>();
uint32_t Src1 = *GetSrc<uint32_t*>(Data->SSAData, Op->Src1);
uint8_t *Src2 = GetSrc<uint8_t*>(Data->SSAData, Op->Src2);
uint32_t Tmp{};
switch (Op->SrcSize) {
case 1:
Tmp = CRC32::crc32cb(Src1, *(uint8_t*)Src2);
break;
case 2:
Tmp = CRC32::crc32ch(Src1, *(uint16_t*)Src2);
break;
case 4:
Tmp = CRC32::crc32cw(Src1, *(uint32_t*)Src2);
break;
case 8:
Tmp = CRC32::crc32cx(Src1, *(uint64_t*)Src2);
break;
default:
LOGMAN_MSG_A_FMT("Unknown CRC32C size: {}", Op->SrcSize);
break;
}
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(PCLMUL) {
auto Op = IROp->C<IR::IROp_PCLMUL>();
const auto Selector = Op->Selector;
auto* Dst = GetDest<uint64_t*>(Data->SSAData, Node);
auto* Src1 = GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
auto* Src2 = GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
const uint64_t TMP1 = (Selector & 0x01) == 0 ? Src1[0] : Src1[1];
const uint64_t TMP2 = (Selector & 0x10) == 0 ? Src2[0] : Src2[1];
const auto make_lo = [](uint64_t lhs, uint64_t rhs) {
uint64_t result = 0;
for (size_t i = 0; i < 64; i++) {
if ((lhs & (1ULL << i)) != 0) {
result ^= rhs << i;
}
}
return result;
};
const auto make_hi = [](uint64_t lhs, uint64_t rhs) {
uint64_t result = 0;
for (size_t i = 1; i < 64; i++) {
if ((lhs & (1ULL << i)) != 0) {
result ^= rhs >> (64 - i);
}
}
return result;
};
Dst[0] = make_lo(TMP1, TMP2);
Dst[1] = make_hi(TMP1, TMP2);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,423 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include "F80Ops.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(F80LOADFCW) {
FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle(*GetSrc<uint16_t*>(Data->SSAData, IROp->Args[0]));
}
DEF_OP(F80ADD) {
auto Op = IROp->C<IR::IROp_F80Add>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FADD(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SUB) {
auto Op = IROp->C<IR::IROp_F80Sub>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FSUB(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80MUL) {
auto Op = IROp->C<IR::IROp_F80Mul>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FMUL(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80DIV) {
auto Op = IROp->C<IR::IROp_F80Div>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FDIV(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80FYL2X) {
auto Op = IROp->C<IR::IROp_F80FYL2X>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FYL2X(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80ATAN) {
auto Op = IROp->C<IR::IROp_F80ATAN>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FATAN(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80FPREM1) {
auto Op = IROp->C<IR::IROp_F80FPREM1>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FREM1(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80FPREM) {
auto Op = IROp->C<IR::IROp_F80FPREM>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FREM(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SCALE) {
auto Op = IROp->C<IR::IROp_F80SCALE>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FSCALE(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80CVT) {
auto Op = IROp->C<IR::IROp_F80CVT>();
const uint8_t OpSize = IROp->Size;
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
switch (OpSize) {
case 4: {
float Tmp = Src;
memcpy(GDP, &Tmp, OpSize);
break;
}
case 8: {
double Tmp = Src;
memcpy(GDP, &Tmp, OpSize);
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
}
DEF_OP(F80CVTINT) {
auto Op = IROp->C<IR::IROp_F80CVTInt>();
const uint8_t OpSize = IROp->Size;
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
switch (OpSize) {
case 2: {
int16_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2)(Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
case 4: {
int32_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4)(Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
case 8: {
int64_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8)(Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
}
DEF_OP(F80CVTTO) {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch (Op->SrcSize) {
case 4: {
float Src = *GetSrc<float *>(Data->SSAData, Op->X80Src);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
case 8: {
double Src = *GetSrc<double *>(Data->SSAData, Op->X80Src);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
}
}
DEF_OP(F80CVTTOINT) {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->SrcSize) {
case 2: {
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Src);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
case 4: {
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Src);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
}
}
DEF_OP(F80ROUND) {
auto Op = IROp->C<IR::IROp_F80Round>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FRNDINT(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80F2XM1) {
auto Op = IROp->C<IR::IROp_F80F2XM1>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::F2XM1(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80TAN) {
auto Op = IROp->C<IR::IROp_F80TAN>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FTAN(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SQRT) {
auto Op = IROp->C<IR::IROp_F80SQRT>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FSQRT(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SIN) {
auto Op = IROp->C<IR::IROp_F80SIN>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FSIN(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80COS) {
auto Op = IROp->C<IR::IROp_F80COS>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FCOS(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80XTRACT_EXP) {
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FXTRACT_EXP(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80XTRACT_SIG) {
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FXTRACT_SIG(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80CMP) {
auto Op = IROp->C<IR::IROp_F80Cmp>();
uint32_t ResultFlags{};
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
bool eq, lt, nan;
X80SoftFloat::FCMP(Src1, Src2, &eq, &lt, &nan);
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
nan) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ) &&
eq) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
GD = ResultFlags;
}
DEF_OP(F80BCDLOAD) {
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->X80Src);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
// Each 4bit split is a digit
// Only 0-9 is supported, A-F results in undefined data
// | 4 bit | 4 bit |
// | 10s place | 1s place |
// EG 0x48 = 48
// EG 0x4847 = 4847
// This gives us an 18digit value encoded in BCD
// The last byte lets us know if it negative or not
for (size_t i = 0; i < 9; ++i) {
uint8_t Digit = Src1[8 - i];
// First shift our last value over
BCD *= 100;
// Add the tens place digit
BCD += (Digit >> 4) * 10;
// Add the ones place digit
BCD += Digit & 0xF;
}
// Set negative flag once converted to x87
bool Negative = Src1[9] & 0x80;
X80SoftFloat Tmp;
Tmp = BCD;
Tmp.Sign = Negative;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80BCDSTORE) {
auto Op = IROp->C<IR::IROp_F80BCDStore>();
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src));
bool Negative = Src1.Sign;
// Clear the Sign bit
Src1.Sign = 0;
uint64_t Tmp = Src1;
uint8_t BCD[10]{};
for (size_t i = 0; i < 9; ++i) {
if (Tmp == 0) {
// Nothing left? Just leave
break;
}
// Extract the lower 100 values
uint8_t Digit = Tmp % 100;
// Now divide it for the next iteration
Tmp /= 100;
uint8_t UpperNibble = Digit / 10;
uint8_t LowerNibble = Digit % 10;
// Now store the BCD
BCD[i] = (UpperNibble << 4) | LowerNibble;
}
// Set negative flag once converted to x87
BCD[9] = Negative ? 0x80 : 0;
memcpy(GDP, BCD, 10);
}
DEF_OP(F64SIN) {
auto Op = IROp->C<IR::IROp_F64SIN>();
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Tmp = sin(Src);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64COS) {
auto Op = IROp->C<IR::IROp_F64COS>();
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Tmp = cos(Src);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64TAN) {
auto Op = IROp->C<IR::IROp_F64TAN>();
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Tmp = tan(Src);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64F2XM1) {
auto Op = IROp->C<IR::IROp_F64F2XM1>();
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Tmp = exp2(Src) - 1.0;
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64ATAN) {
auto Op = IROp->C<IR::IROp_F64ATAN>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double Tmp = atan2(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64FPREM) {
auto Op = IROp->C<IR::IROp_F64FPREM>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double Tmp = fmod(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64FPREM1) {
auto Op = IROp->C<IR::IROp_F64FPREM1>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double Tmp = remainder(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64FYL2X) {
auto Op = IROp->C<IR::IROp_F64FYL2X>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double Tmp = Src2 * log2(Src1);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64SCALE) {
auto Op = IROp->C<IR::IROp_F64SCALE>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double trunc = (double)(int64_t)(Src2); //truncate
const double Tmp = Src1 * exp2(trunc);
memcpy(GDP, &Tmp, sizeof(double));
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,399 +0,0 @@
#pragma once
#include "Common/SoftFloat.h"
#include "Common/SoftFloat-3e/softfloat.h"
#include <FEXCore/IR/IR.h>
namespace FEXCore::CPU {
template<IR::IROps Op>
struct OpHandlers {
};
template<>
struct OpHandlers<IR::OP_F80CVTTO> {
static X80SoftFloat handle4(float src) {
return src;
}
static X80SoftFloat handle8(double src) {
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80CMP> {
template<uint32_t Flags>
static uint64_t handle(X80SoftFloat Src1, X80SoftFloat Src2) {
bool eq, lt, nan;
uint64_t ResultFlags = 0;
X80SoftFloat::FCMP(Src1, Src2, &eq, &lt, &nan);
if (Flags & (1 << IR::FCMP_FLAG_LT) &&
lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
nan) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
if (Flags & (1 << IR::FCMP_FLAG_EQ) &&
eq) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
return ResultFlags;
}
};
template<>
struct OpHandlers<IR::OP_F80CVT> {
static float handle4(X80SoftFloat src) {
return src;
}
static double handle8(X80SoftFloat src) {
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80CVTINT> {
static int16_t handle2(X80SoftFloat src) {
return src;
}
static int32_t handle4(X80SoftFloat src) {
return src;
}
static int64_t handle8(X80SoftFloat src) {
return src;
}
static int16_t handle2t(X80SoftFloat src) {
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
if (rv > INT16_MAX) {
return INT16_MAX;
} else if (rv < INT16_MIN) {
return INT16_MIN;
} else {
return rv;
}
}
static int32_t handle4t(X80SoftFloat src) {
return extF80_to_i32(src, softfloat_round_minMag, false);
}
static int64_t handle8t(X80SoftFloat src) {
return extF80_to_i64(src, softfloat_round_minMag, false);
}
};
template<>
struct OpHandlers<IR::OP_F80CVTTOINT> {
static X80SoftFloat handle2(int16_t src) {
return src;
}
static X80SoftFloat handle4(int32_t src) {
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80ROUND> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FRNDINT(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80F2XM1> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::F2XM1(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80TAN> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FTAN(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80SQRT> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FSQRT(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80SIN> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FSIN(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80COS> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FCOS(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FXTRACT_EXP(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FXTRACT_SIG(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80ADD> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FADD(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80SUB> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FSUB(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80MUL> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FMUL(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80DIV> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FDIV(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FYL2X> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FYL2X(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80ATAN> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FATAN(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FPREM1> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FREM1(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FPREM> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FREM(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80SCALE> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FSCALE(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F64SIN> {
static double handle(double src) {
return sin(src);
}
};
template<>
struct OpHandlers<IR::OP_F64COS> {
static double handle(double src) {
return cos(src);
}
};
template<>
struct OpHandlers<IR::OP_F64TAN> {
static double handle(double src) {
return tan(src);
}
};
template<>
struct OpHandlers<IR::OP_F64F2XM1> {
static double handle(double src) {
return exp2(src) - 1.0;
}
};
template<>
struct OpHandlers<IR::OP_F64ATAN> {
static double handle(double src1, double src2) {
return atan2(src1, src2);
}
};
template<>
struct OpHandlers<IR::OP_F64FPREM> {
static double handle(double src1, double src2) {
return fmod(src1, src2);
}
};
template<>
struct OpHandlers<IR::OP_F64FPREM1> {
static double handle(double src1, double src2) {
return remainder(src1, src2);
}
};
template<>
struct OpHandlers<IR::OP_F64FYL2X> {
static double handle(double src1, double src2) {
return src2 * log2(src1);
}
};
template<>
struct OpHandlers<IR::OP_F64SCALE> {
static double handle(double src1, double src2) {
double trunc = (double)(int64_t)(src2); //truncate
return src1 * exp2(trunc);
}
};
template<>
struct OpHandlers<IR::OP_F80BCDSTORE> {
static X80SoftFloat handle(X80SoftFloat Src1) {
bool Negative = Src1.Sign;
Src1 = X80SoftFloat::FRNDINT(Src1);
// Clear the Sign bit
Src1.Sign = 0;
uint64_t Tmp = Src1;
X80SoftFloat Rv;
uint8_t *BCD = reinterpret_cast<uint8_t*>(&Rv);
memset(BCD, 0, 10);
for (size_t i = 0; i < 9; ++i) {
if (Tmp == 0) {
// Nothing left? Just leave
break;
}
// Extract the lower 100 values
uint8_t Digit = Tmp % 100;
// Now divide it for the next iteration
Tmp /= 100;
uint8_t UpperNibble = Digit / 10;
uint8_t LowerNibble = Digit % 10;
// Now store the BCD
BCD[i] = (UpperNibble << 4) | LowerNibble;
}
// Set negative flag once converted to x87
BCD[9] = Negative ? 0x80 : 0;
return Rv;
}
};
template<>
struct OpHandlers<IR::OP_F80BCDLOAD> {
static X80SoftFloat handle(X80SoftFloat Src) {
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
// Each 4bit split is a digit
// Only 0-9 is supported, A-F results in undefined data
// | 4 bit | 4 bit |
// | 10s place | 1s place |
// EG 0x48 = 48
// EG 0x4847 = 4847
// This gives us an 18digit value encoded in BCD
// The last byte lets us know if it negative or not
for (size_t i = 0; i < 9; ++i) {
uint8_t Digit = Src1[8 - i];
// First shift our last value over
BCD *= 100;
// Add the tens place digit
BCD += (Digit >> 4) * 10;
// Add the ones place digit
BCD += Digit & 0xF;
}
// Set negative flag once converted to x87
bool Negative = Src1[9] & 0x80;
X80SoftFloat Tmp;
Tmp = BCD;
Tmp.Sign = Negative;
return Tmp;
}
};
template<>
struct OpHandlers<IR::OP_F80LOADFCW> {
static void handle(uint16_t NewFCW) {
auto PC = (NewFCW >> 8) & 3;
switch(PC) {
case 0: extF80_roundingPrecision = 32; break;
case 2: extF80_roundingPrecision = 64; break;
case 3: extF80_roundingPrecision = 80; break;
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
}
auto RC = (NewFCW >> 10) & 3;
switch(RC) {
case 0:
softfloat_roundingMode = softfloat_round_near_even;
break;
case 1:
softfloat_roundingMode = softfloat_round_min;
break;
case 2:
softfloat_roundingMode = softfloat_round_max;
break;
case 3:
softfloat_roundingMode = softfloat_round_minMag;
break;
}
}
};
}
@@ -1,21 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(GetHostFlag) {
auto Op = IROp->C<IR::IROp_GetHostFlag>();
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Value) >> Op->Flag) & 1;
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,5 +1,6 @@
#pragma once
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
@@ -8,7 +9,6 @@
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::CPU {
class Dispatcher;
class X86DispatchGenerator;
class Arm64DispatchGenerator;
@@ -21,35 +21,32 @@ using DestMapType = std::vector<uint32_t>;
class InterpreterCore final : public CPUBackend {
public:
explicit InterpreterCore(Dispatcher *Dispatch,
FEXCore::Core::InternalThreadState *Thread);
explicit InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
~InterpreterCore() override;
std::string GetName() override { return "Interpreter"; }
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]] void *CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
bool NeedsOpDispatch() override { return true; }
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
void ClearCache() override;
bool HandleSIGBUS(int Signal, void *info, void *ucontext);
private:
size_t BufferUsed;
Dispatcher *Dispatch;
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *State;
uint32_t AllocateTmpSpace(size_t Size);
template<typename Res>
Res GetDest(void* SSAData, IR::OrderedNodeWrapper Op);
template<typename Res>
Res GetSrc(void* SSAData, IR::OrderedNodeWrapper Src);
Dispatcher *Dispatcher{};
};
template<typename T>
T AtomicCompareAndSwap(T expected, T desired, T *addr);
uint8_t AtomicFetchNeg(uint8_t *Addr);
uint16_t AtomicFetchNeg(uint16_t *Addr);
uint32_t AtomicFetchNeg(uint32_t *Addr);
uint64_t AtomicFetchNeg(uint64_t *Addr);
} // namespace FEXCore::CPU
}
@@ -1,109 +1,128 @@
#include "Common/MathUtils.h"
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/Arm64.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/DebugData.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <memory>
#include <signal.h>
#include <stdint.h>
#include <utility>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include "Interface/HLE/Thunks/Thunks.h"
#include <atomic>
#include <cmath>
#include <limits>
#include <vector>
#include "InterpreterOps.h"
#if defined(_M_X86_64)
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
#elif defined(_M_ARM_64)
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
#else
#error missing arch
#endif
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
namespace FEXCore::IR {
class IRListView;
class RegisterAllocationData;
}
namespace FEXCore::CPU {
InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::InternalThreadState *Thread)
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
, Dispatch(Dispatcher)
{
static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
auto Thread = Frame->Thread;
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
auto LocalEntry = Thread->LocalIRCache.find(Thread->CurrentFrame->State.rip);
Interpreter.FragmentExecuter = reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR);
ClearCache();
InterpreterOps::InterpretIR(Thread, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
}
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
bool InterpreterCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
#ifdef _M_ARM_64
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
}, true);
constexpr bool is_arm64 = true;
#else
constexpr bool is_arm64 = false;
#endif
}
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
const auto IRSize = AlignUp(IR->GetInlineSize(), 16);
const auto MaxSize = IRSize + Dispatcher::MaxInterpreterTrampolineSize + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
if ((BufferUsed + MaxSize) > CurrentCodeBuffer->Size) {
ThreadState->CTX->ClearCodeCache(ThreadState);
if constexpr (is_arm64) {
uint32_t *PC = reinterpret_cast<uint32_t*>(ArchHelpers::Context::GetPc(ucontext));
uint32_t Instr = PC[0];
if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
LogMan::Msg::E("Unhandled JIT SIGBUS CASPAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
return false;
}
}
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
LogMan::Msg::E("Unhandled JIT SIGBUS CASAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
return false;
}
}
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
uint8_t Op = (PC[0] >> 12) & 0xF;
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x: PC: %p Instruction: 0x%08x\n", Op, PC, PC[0]);
return false;
}
}
}
return false;
}
const auto BufferStart = CurrentCodeBuffer->Ptr + BufferUsed;
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
: CTX {ctx}
, State {Thread} {
// Grab our space for temporary data
auto DestBuffer = BufferStart;
if (!CompileThread &&
CTX->Config.Core == FEXCore::Config::CONFIG_INTERPRETER) {
CreateAsmDispatch(ctx, Thread);
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
});
if (GDBEnabled) {
const auto GDBSize = Dispatch->GenerateGDBPauseCheck(DestBuffer, Entry);
DestBuffer += GDBSize;
BufferUsed += GDBSize;
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->HandleSIGBUS(Signal, info, ucontext);
});
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
};
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
}
}
const auto TrampolineSize = Dispatch->GenerateInterpreterTrampoline(DestBuffer);
DestBuffer += TrampolineSize;
BufferUsed += TrampolineSize;
IR->Serialize(DestBuffer);
DestBuffer += IRSize;
BufferUsed += IRSize;
return BufferStart;
}
void InterpreterCore::ClearCache() {
// Calling this one is needed to setup the initial CurrentCodeBuffer
[[maybe_unused]] auto CodeBuffer = GetEmptyCodeBuffer();
BufferUsed = 0;
InterpreterCore::~InterpreterCore() {
delete Dispatcher;
}
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
return std::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
void *InterpreterCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
return reinterpret_cast<void*>(InterpreterExecution);
}
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
InterpreterCore::InitializeSignalHandlers(CTX);
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
return new InterpreterCore(ctx, Thread, CompileThread);
}
CPUBackendFeatures GetInterpreterBackendFeatures() {
return CPUBackendFeatures { };
}
}
@@ -1,7 +1,5 @@
#pragma once
#include <memory>
namespace FEXCore::Context {
struct Context;
}
@@ -12,11 +10,7 @@ namespace FEXCore::Core {
namespace FEXCore::CPU {
class CPUBackend;
struct DispatcherConfig;
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread);
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
CPUBackendFeatures GetInterpreterBackendFeatures();
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
} // namespace FEXCore::CPU
}
@@ -1,184 +0,0 @@
#pragma once
#include <FEXCore/IR/IR.h>
#define GD *GetDest<uint64_t*>(Data->SSAData, Node)
#define GDP GetDest<void*>(Data->SSAData, Node)
#define DO_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(GDP); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
*Dst_d = func(*Src1_d, *Src2_d); \
break; \
}
#define DO_SCALAR_COMPARE_OP(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
Dst_d[0] = func(Src1_d[0], Src2_d[0]); \
break; \
}
#define DO_VECTOR_COMPARE_OP(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
} \
break; \
}
#define DO_VECTOR_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
} \
break; \
}
#define DO_VECTOR_PAIR_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i*2], Src1_d[i*2 + 1]); \
Dst_d[i+Elements] = func(Src2_d[i*2], Src2_d[i*2 + 1]); \
} \
break; \
}
#define DO_VECTOR_SCALAR_OP(size, type, func)\
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], *Src2_d); \
} \
break; \
}
#define DO_VECTOR_0SRC_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(); \
} \
break; \
}
#define DO_VECTOR_1SRC_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src_d[i]); \
} \
break; \
}
#define DO_VECTOR_REDUCE_1SRC_OP(size, type, func, start_val) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type*>(Src); \
type begin = start_val; \
for (uint8_t i = 0; i < Elements; ++i) { \
begin = func(begin, Src_d[i]); \
} \
Dst_d[0] = begin; \
break; \
}
#define DO_VECTOR_SAT_OP(size, type, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], Src2_d[i], min, max); \
} \
break; \
}
#define DO_VECTOR_1SRC_2TYPE_OP(size, type, type2, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func(Src_d[i], min, max); \
} \
break; \
}
#define DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(type, type2, func, min, max) \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func(Src_d[i], min, max); \
}
#define DO_VECTOR_1SRC_2TYPE_OP_TOP(size, type, type2, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src2); \
memcpy(Dst_d, Src1, Elements * sizeof(type2));\
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i+Elements] = (type)func(Src_d[i], min, max); \
} \
break; \
}
#define DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(size, type, type2, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func(Src_d[i+Elements], min, max); \
} \
break; \
}
#define DO_VECTOR_2SRC_2TYPE_OP(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func((type)Src1_d[i], (type)Src2_d[i]); \
} \
break; \
}
#define DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func((type)Src1_d[i+Elements], (type)Src2_d[i+Elements]); \
} \
break; \
}
struct InterpVector256 {
__uint128_t Lower;
__uint128_t Upper;
};
template<typename Res>
Res GetDest(void* SSAData, FEXCore::IR::OrderedNodeWrapper Op) {
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.ID().Value];
return reinterpret_cast<Res>(DstPtr);
}
template<typename Res>
Res GetDest(void* SSAData, FEXCore::IR::NodeID Op) {
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.Value];
return reinterpret_cast<Res>(DstPtr);
}
template<typename Res>
Res GetSrc(void* SSAData, FEXCore::IR::OrderedNodeWrapper Src) {
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Src.ID().Value];
return reinterpret_cast<Res>(DstPtr);
}
@@ -1,313 +0,0 @@
#include "FEXCore/Core/CoreState.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/F80Ops.h"
#include <cstddef>
#include <cstdint>
namespace FEXCore::CPU {
template<typename R, typename... Args>
static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_UNKNOWN, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F32, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_VOID_U16, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I32, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F32_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F64_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I16_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I32_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_F80_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F80_F80, (void*)fn, HandlerIndex};
}
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
Info[Core::OPINDEX_F80LOADFCW] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW).fn);
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4).fn);
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8).fn);
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4).fn);
Info[Core::OPINDEX_F80CVT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8).fn);
Info[Core::OPINDEX_F80CVTINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2).fn);
Info[Core::OPINDEX_F80CVTINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4).fn);
Info[Core::OPINDEX_F80CVTINT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8).fn);
Info[Core::OPINDEX_F80CVTINT_TRUNC2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2).fn);
Info[Core::OPINDEX_F80CVTINT_TRUNC4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4).fn);
Info[Core::OPINDEX_F80CVTINT_TRUNC8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8).fn);
Info[Core::OPINDEX_F80CMP_0] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>, Core::OPINDEX_F80CMP_0).fn);
Info[Core::OPINDEX_F80CMP_1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>, Core::OPINDEX_F80CMP_1).fn);
Info[Core::OPINDEX_F80CMP_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>, Core::OPINDEX_F80CMP_2).fn);
Info[Core::OPINDEX_F80CMP_3] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>, Core::OPINDEX_F80CMP_3).fn);
Info[Core::OPINDEX_F80CMP_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>, Core::OPINDEX_F80CMP_4).fn);
Info[Core::OPINDEX_F80CMP_5] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>, Core::OPINDEX_F80CMP_5).fn);
Info[Core::OPINDEX_F80CMP_6] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>, Core::OPINDEX_F80CMP_6).fn);
Info[Core::OPINDEX_F80CMP_7] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>, Core::OPINDEX_F80CMP_7).fn);
Info[Core::OPINDEX_F80CVTTOINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2).fn);
Info[Core::OPINDEX_F80CVTTOINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4).fn);
// Unary
Info[Core::OPINDEX_F80ROUND] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ROUND>::handle, Core::OPINDEX_F80ROUND).fn);
Info[Core::OPINDEX_F80F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80F2XM1>::handle, Core::OPINDEX_F80F2XM1).fn);
Info[Core::OPINDEX_F80TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80TAN>::handle, Core::OPINDEX_F80TAN).fn);
Info[Core::OPINDEX_F80SQRT] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SQRT>::handle, Core::OPINDEX_F80SQRT).fn);
Info[Core::OPINDEX_F80SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SIN>::handle, Core::OPINDEX_F80SIN).fn);
Info[Core::OPINDEX_F80COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80COS>::handle, Core::OPINDEX_F80COS).fn);
Info[Core::OPINDEX_F80XTRACT_EXP] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle, Core::OPINDEX_F80XTRACT_EXP).fn);
Info[Core::OPINDEX_F80XTRACT_SIG] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle, Core::OPINDEX_F80XTRACT_SIG).fn);
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle, Core::OPINDEX_F80BCDSTORE).fn);
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle, Core::OPINDEX_F80BCDLOAD).fn);
// Binary
Info[Core::OPINDEX_F80ADD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ADD>::handle, Core::OPINDEX_F80ADD).fn);
Info[Core::OPINDEX_F80SUB] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SUB>::handle, Core::OPINDEX_F80SUB).fn);
Info[Core::OPINDEX_F80MUL] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80MUL>::handle, Core::OPINDEX_F80MUL).fn);
Info[Core::OPINDEX_F80DIV] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80DIV>::handle, Core::OPINDEX_F80DIV).fn);
Info[Core::OPINDEX_F80FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2X>::handle, Core::OPINDEX_F80FYL2X).fn);
Info[Core::OPINDEX_F80ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ATAN>::handle, Core::OPINDEX_F80ATAN).fn);
Info[Core::OPINDEX_F80FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM1>::handle, Core::OPINDEX_F80FPREM1).fn);
Info[Core::OPINDEX_F80FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM>::handle, Core::OPINDEX_F80FPREM).fn);
Info[Core::OPINDEX_F80SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle, Core::OPINDEX_F80SCALE).fn);
// Double Precision
Info[Core::OPINDEX_F64SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle, Core::OPINDEX_F64SIN).fn);
Info[Core::OPINDEX_F64COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle, Core::OPINDEX_F64COS).fn);
Info[Core::OPINDEX_F64TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle, Core::OPINDEX_F64TAN).fn);
Info[Core::OPINDEX_F64ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle, Core::OPINDEX_F64ATAN).fn);
Info[Core::OPINDEX_F64F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle, Core::OPINDEX_F64F2XM1).fn);
Info[Core::OPINDEX_F64FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle, Core::OPINDEX_F64FYL2X).fn);
Info[Core::OPINDEX_F64FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle, Core::OPINDEX_F64FPREM).fn);
Info[Core::OPINDEX_F64FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle, Core::OPINDEX_F64FPREM1).fn);
Info[Core::OPINDEX_F64SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle, Core::OPINDEX_F64SCALE).fn);
}
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
uint8_t OpSize = IROp->Size;
switch(IROp->Op) {
case IR::OP_F80LOADFCW: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW);
return true;
}
case IR::OP_F80CVTTO: {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch (Op->SrcSize) {
case 4: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4);
return true;
}
case 8: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8);
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CVT: {
switch (OpSize) {
case 4: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4);
return true;
}
case 8: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8);
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CVTINT: {
auto Op = IROp->C<IR::IROp_F80CVTInt>();
switch (OpSize) {
case 2: {
if (Op->Truncate) {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2);
}
else {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2);
}
return true;
}
case 4: {
if (Op->Truncate) {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4);
}
else {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4);
}
return true;
}
case 8: {
if (Op->Truncate) {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8);
}
else {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8);
}
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CMP: {
auto Op = IROp->C<IR::IROp_F80Cmp>();
static constexpr std::array handlers{
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
};
*Info = GetFallbackInfo(handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags));
return true;
}
case IR::OP_F80CVTTOINT: {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->SrcSize) {
case 2: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2);
return true;
}
case 4: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4);
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
#define COMMON_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP); \
return true; \
}
#define COMMON_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
return true; \
}
// Unary
COMMON_X87_OP(ROUND)
COMMON_X87_OP(F2XM1)
COMMON_X87_OP(TAN)
COMMON_X87_OP(SQRT)
COMMON_X87_OP(SIN)
COMMON_X87_OP(COS)
COMMON_X87_OP(XTRACT_EXP)
COMMON_X87_OP(XTRACT_SIG)
COMMON_X87_OP(BCDSTORE)
COMMON_X87_OP(BCDLOAD)
// Binary
COMMON_X87_OP(ADD)
COMMON_X87_OP(SUB)
COMMON_X87_OP(MUL)
COMMON_X87_OP(DIV)
COMMON_X87_OP(FYL2X)
COMMON_X87_OP(ATAN)
COMMON_X87_OP(FPREM1)
COMMON_X87_OP(FPREM)
COMMON_X87_OP(SCALE)
// Double Precision Unary
COMMON_F64_OP(F2XM1)
COMMON_F64_OP(TAN)
COMMON_F64_OP(SIN)
COMMON_F64_OP(COS)
// Double Precision Binary
COMMON_F64_OP(FYL2X)
COMMON_F64_OP(ATAN)
COMMON_F64_OP(FPREM1)
COMMON_F64_OP(FPREM)
COMMON_F64_OP(SCALE)
default:
break;
}
return false;
}
}
File diff suppressed because it is too large. Load diff
@@ -1,17 +1,9 @@
#pragma once
#include <stdint.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::Core {
struct InternalThreadState;
}
namespace FEXCore::IR {
class IRListView;
struct IROp_Header;
}
namespace FEXCore::Core{
@@ -28,8 +20,6 @@ namespace FEXCore::CPU {
FABI_F80_I32,
FABI_F32_F80,
FABI_F64_F80,
FABI_F64_F64,
FABI_F64_F64_F64,
FABI_I16_F80,
FABI_I32_F80,
FABI_I64_F80,
@@ -41,372 +31,12 @@ namespace FEXCore::CPU {
struct FallbackInfo {
FallbackABI ABI;
void *fn;
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
};
class InterpreterOps {
public:
static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR);
static void FillFallbackIndexPointers(uint64_t *Info);
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
struct IROpData {
FEXCore::Core::InternalThreadState *State{};
uint64_t CurrentEntry{};
FEXCore::IR::IRListView const *CurrentIR{};
volatile void *StackEntry{};
void *SSAData{};
struct {
bool Quit;
bool Redo;
} BlockResults{};
IR::NodeIterator BlockIterator{0, 0};
};
#define DEF_OP(x) static void Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
///< Unhandled handler
DEF_OP(Unhandled);
///< No-op Handler
DEF_OP(NoOp);
///< ALU Ops
DEF_OP(TruncElementPair);
DEF_OP(Constant);
DEF_OP(EntrypointOffset);
DEF_OP(InlineConstant);
DEF_OP(InlineEntrypointOffset);
DEF_OP(CycleCounter);
DEF_OP(Add);
DEF_OP(Sub);
DEF_OP(Neg);
DEF_OP(Mul);
DEF_OP(UMul);
DEF_OP(Div);
DEF_OP(UDiv);
DEF_OP(Rem);
DEF_OP(URem);
DEF_OP(MulH);
DEF_OP(UMulH);
DEF_OP(Or);
DEF_OP(And);
DEF_OP(Andn);
DEF_OP(Xor);
DEF_OP(Lshl);
DEF_OP(Lshr);
DEF_OP(Ashr);
DEF_OP(Rol);
DEF_OP(Ror);
DEF_OP(Extr);
DEF_OP(PDep);
DEF_OP(PExt);
DEF_OP(LDiv);
DEF_OP(LUDiv);
DEF_OP(LRem);
DEF_OP(LURem);
DEF_OP(Zext);
DEF_OP(Not);
DEF_OP(Popcount);
DEF_OP(FindLSB);
DEF_OP(FindMSB);
DEF_OP(FindTrailingZeros);
DEF_OP(CountLeadingZeroes);
DEF_OP(Rev);
DEF_OP(Bfi);
DEF_OP(Bfe);
DEF_OP(Sbfe);
DEF_OP(Select);
DEF_OP(VExtractToGPR);
DEF_OP(Float_ToGPR_ZU);
DEF_OP(Float_ToGPR_ZS);
DEF_OP(Float_ToGPR_S);
DEF_OP(FCmp);
///< Atomic ops
DEF_OP(CASPair);
DEF_OP(CAS);
DEF_OP(AtomicAdd);
DEF_OP(AtomicSub);
DEF_OP(AtomicAnd);
DEF_OP(AtomicOr);
DEF_OP(AtomicXor);
DEF_OP(AtomicSwap);
DEF_OP(AtomicFetchAdd);
DEF_OP(AtomicFetchSub);
DEF_OP(AtomicFetchAnd);
DEF_OP(AtomicFetchOr);
DEF_OP(AtomicFetchXor);
DEF_OP(AtomicFetchNeg);
///< Branch ops
DEF_OP(SignalReturn);
DEF_OP(CallbackReturn);
DEF_OP(ExitFunction);
DEF_OP(Jump);
DEF_OP(CondJump);
DEF_OP(Syscall);
DEF_OP(InlineSyscall);
DEF_OP(Thunk);
DEF_OP(ValidateCode);
DEF_OP(ThreadRemoveCodeEntry);
DEF_OP(CPUID);
///< Conversion ops
DEF_OP(VInsGPR);
DEF_OP(VCastFromGPR);
DEF_OP(Float_FromGPR_S);
DEF_OP(Float_FToF);
DEF_OP(Vector_SToF);
DEF_OP(Vector_FToZS);
DEF_OP(Vector_FToS);
DEF_OP(Vector_FToF);
DEF_OP(Vector_FToI);
///< Flag ops
DEF_OP(GetHostFlag);
///< Memory ops
DEF_OP(LoadContext);
DEF_OP(StoreContext);
DEF_OP(LoadRegister);
DEF_OP(StoreRegister);
DEF_OP(LoadContextIndexed);
DEF_OP(StoreContextIndexed);
DEF_OP(SpillRegister);
DEF_OP(FillRegister);
DEF_OP(LoadFlag);
DEF_OP(StoreFlag);
DEF_OP(LoadMem);
DEF_OP(StoreMem);
DEF_OP(VLoadMemElement);
DEF_OP(VStoreMemElement);
DEF_OP(CacheLineClear);
DEF_OP(CacheLineZero);
///< Misc ops
DEF_OP(EndBlock);
DEF_OP(Fence);
DEF_OP(Break);
DEF_OP(Phi);
DEF_OP(PhiValue);
DEF_OP(Print);
DEF_OP(GetRoundingMode);
DEF_OP(SetRoundingMode);
DEF_OP(ProcessorID);
DEF_OP(RDRAND);
DEF_OP(Yield);
///< Move ops
DEF_OP(ExtractElementPair);
DEF_OP(CreateElementPair);
DEF_OP(Mov);
///< Vector ops
DEF_OP(VectorZero);
DEF_OP(VectorImm);
DEF_OP(VMov);
DEF_OP(VAnd);
DEF_OP(VBic);
DEF_OP(VOr);
DEF_OP(VXor);
DEF_OP(VAdd);
DEF_OP(VSub);
DEF_OP(VUQAdd);
DEF_OP(VUQSub);
DEF_OP(VSQAdd);
DEF_OP(VSQSub);
DEF_OP(VAddP);
DEF_OP(VAddV);
DEF_OP(VUMinV);
DEF_OP(VURAvg);
DEF_OP(VAbs);
DEF_OP(VPopcount);
DEF_OP(VFAdd);
DEF_OP(VFAddP);
DEF_OP(VFSub);
DEF_OP(VFMul);
DEF_OP(VFDiv);
DEF_OP(VFMin);
DEF_OP(VFMax);
DEF_OP(VFRecp);
DEF_OP(VFSqrt);
DEF_OP(VFRSqrt);
DEF_OP(VNeg);
DEF_OP(VFNeg);
DEF_OP(VNot);
DEF_OP(VUMin);
DEF_OP(VSMin);
DEF_OP(VUMax);
DEF_OP(VSMax);
DEF_OP(VZip);
DEF_OP(VUnZip);
DEF_OP(VBSL);
DEF_OP(VCMPEQ);
DEF_OP(VCMPEQZ);
DEF_OP(VCMPGT);
DEF_OP(VCMPGTZ);
DEF_OP(VCMPLTZ);
DEF_OP(VFCMPEQ);
DEF_OP(VFCMPNEQ);
DEF_OP(VFCMPLT);
DEF_OP(VFCMPGT);
DEF_OP(VFCMPLE);
DEF_OP(VFCMPORD);
DEF_OP(VFCMPUNO);
DEF_OP(VUShl);
DEF_OP(VUShr);
DEF_OP(VSShr);
DEF_OP(VUShlS);
DEF_OP(VUShrS);
DEF_OP(VSShrS);
DEF_OP(VInsElement);
DEF_OP(VDupElement);
DEF_OP(VExtr);
DEF_OP(VSLI);
DEF_OP(VSRI);
DEF_OP(VUShrI);
DEF_OP(VSShrI);
DEF_OP(VShlI);
DEF_OP(VUShrNI);
DEF_OP(VUShrNI2);
DEF_OP(VSXTL);
DEF_OP(VSXTL2);
DEF_OP(VUXTL);
DEF_OP(VUXTL2);
DEF_OP(VSQXTN);
DEF_OP(VSQXTN2);
DEF_OP(VSQXTUN);
DEF_OP(VSQXTUN2);
DEF_OP(VUMul);
DEF_OP(VUMull);
DEF_OP(VSMul);
DEF_OP(VSMull);
DEF_OP(VUMull2);
DEF_OP(VSMull2);
DEF_OP(VUABDL);
DEF_OP(VTBL1);
DEF_OP(VRev64);
///< Encryption ops
DEF_OP(AESImc);
DEF_OP(AESEnc);
DEF_OP(AESEncLast);
DEF_OP(AESDec);
DEF_OP(AESDecLast);
DEF_OP(AESKeyGenAssist);
DEF_OP(CRC32);
DEF_OP(PCLMUL);
///< F80 ops
DEF_OP(F80LOADFCW);
DEF_OP(F80ADD);
DEF_OP(F80SUB);
DEF_OP(F80MUL);
DEF_OP(F80DIV);
DEF_OP(F80FYL2X);
DEF_OP(F80ATAN);
DEF_OP(F80FPREM1);
DEF_OP(F80FPREM);
DEF_OP(F80SCALE);
DEF_OP(F80CVT);
DEF_OP(F80CVTINT);
DEF_OP(F80CVTTO);
DEF_OP(F80CVTTOINT);
DEF_OP(F80ROUND);
DEF_OP(F80F2XM1);
DEF_OP(F80TAN);
DEF_OP(F80SQRT);
DEF_OP(F80SIN);
DEF_OP(F80COS);
DEF_OP(F80XTRACT_EXP);
DEF_OP(F80XTRACT_SIG);
DEF_OP(F80CMP);
DEF_OP(F80BCDLOAD);
DEF_OP(F80BCDSTORE);
//< F64 ops
DEF_OP(F64SIN);
DEF_OP(F64COS);
DEF_OP(F64TAN);
DEF_OP(F64F2XM1);
DEF_OP(F64ATAN);
DEF_OP(F64FPREM);
DEF_OP(F64FPREM1);
DEF_OP(F64FYL2X);
DEF_OP(F64SCALE);
#undef DEF_OP
template<typename unsigned_type, typename signed_type, typename float_type>
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
bool CompResult = false;
switch (Cond) {
case FEXCore::IR::COND_EQ:
CompResult = static_cast<unsigned_type>(Src1) == static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_NEQ:
CompResult = static_cast<unsigned_type>(Src1) != static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_SGE:
CompResult = static_cast<signed_type>(Src1) >= static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_SLT:
CompResult = static_cast<signed_type>(Src1) < static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_SGT:
CompResult = static_cast<signed_type>(Src1) > static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_SLE:
CompResult = static_cast<signed_type>(Src1) <= static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_UGE:
CompResult = static_cast<unsigned_type>(Src1) >= static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_ULT:
CompResult = static_cast<unsigned_type>(Src1) < static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_UGT:
CompResult = static_cast<unsigned_type>(Src1) > static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_ULE:
CompResult = static_cast<unsigned_type>(Src1) <= static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_FLU:
CompResult = reinterpret_cast<float_type&>(Src1) < reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FGE:
CompResult = reinterpret_cast<float_type&>(Src1) >= reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FLEU:
CompResult = reinterpret_cast<float_type&>(Src1) <= reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FGT:
CompResult = reinterpret_cast<float_type&>(Src1) > reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FU:
CompResult = (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FNU:
CompResult = !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_MI:
case FEXCore::IR::COND_PL:
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
break;
}
return CompResult;
}
static uint8_t GetOpSize(FEXCore::IR::IRListView const *CurrentIR, IR::OrderedNodeWrapper Node) {
auto IROp = CurrentIR->GetOp<FEXCore::IR::IROp_Header>(Node);
return IROp->Size;
}
};
} // namespace FEXCore::CPU
};
@@ -1,296 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
static inline void CacheLineFlush(char *Addr) {
#ifdef _M_X86_64
__asm volatile (
"clflush (%[Addr]);"
:: [Addr] "r" (Addr)
: "memory");
#else
__builtin___clear_cache(Addr, Addr+64);
#endif
}
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(LoadContext) {
const auto Op = IROp->C<IR::IROp_LoadContext>();
const auto OpSize = IROp->Size;
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Src = ContextPtr + Op->Offset;
#define LOAD_CTX(x, y) \
case x: { \
y const *MemData = reinterpret_cast<y const*>(Src); \
GD = *MemData; \
break; \
}
switch (OpSize) {
LOAD_CTX(1, uint8_t)
LOAD_CTX(2, uint16_t)
LOAD_CTX(4, uint32_t)
LOAD_CTX(8, uint64_t)
case 16:
case 32: {
void const *MemData = reinterpret_cast<void const*>(Src);
memcpy(GDP, MemData, OpSize);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
break;
}
#undef LOAD_CTX
}
DEF_OP(StoreContext) {
const auto Op = IROp->C<IR::IROp_StoreContext>();
const auto OpSize = IROp->Size;
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Dst = ContextPtr + Op->Offset;
void *MemData = reinterpret_cast<void*>(Dst);
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
memcpy(MemData, Src, OpSize);
}
DEF_OP(LoadRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(StoreRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(LoadContextIndexed) {
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
const auto OpSize = IROp->Size;
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Src = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
#define LOAD_CTX(x, y) \
case x: { \
y const *MemData = reinterpret_cast<y const*>(Src); \
GD = *MemData; \
break; \
}
switch (OpSize) {
LOAD_CTX(1, uint8_t)
LOAD_CTX(2, uint16_t)
LOAD_CTX(4, uint32_t)
LOAD_CTX(8, uint64_t)
case 16:
case 32: {
void const *MemData = reinterpret_cast<void const*>(Src);
memcpy(GDP, MemData, OpSize);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
break;
}
#undef LOAD_CTX
}
DEF_OP(StoreContextIndexed) {
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
const auto OpSize = IROp->Size;
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Dst = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
void *MemData = reinterpret_cast<void*>(Dst);
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
memcpy(MemData, Src, OpSize);
}
DEF_OP(SpillRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(FillRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(LoadFlag) {
auto Op = IROp->C<IR::IROp_LoadFlag>();
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
ContextPtr += Op->Flag;
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
GD = *MemData;
}
DEF_OP(StoreFlag) {
auto Op = IROp->C<IR::IROp_StoreFlag>();
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
ContextPtr += Op->Flag;
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
*MemData = Arg;
}
DEF_OP(LoadMem) {
const auto Op = IROp->C<IR::IROp_LoadMem>();
const auto OpSize = IROp->Size;
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
if (!Op->Offset.IsInvalid()) {
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
switch(Op->OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
}
}
memset(GDP, 0, Core::CPUState::XMM_AVX_REG_SIZE);
switch (OpSize) {
case 1: {
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
GD = D->load();
break;
}
case 2: {
auto D = reinterpret_cast<const std::atomic<uint16_t>*>(MemData);
GD = D->load();
break;
}
case 4: {
auto D = reinterpret_cast<const std::atomic<uint32_t>*>(MemData);
GD = D->load();
break;
}
case 8: {
auto D = reinterpret_cast<const std::atomic<uint64_t>*>(MemData);
GD = D->load();
break;
}
default:
memcpy(GDP, MemData, OpSize);
break;
}
}
DEF_OP(StoreMem) {
const auto Op = IROp->C<IR::IROp_StoreMem>();
const auto OpSize = IROp->Size;
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
if (!Op->Offset.IsInvalid()) {
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
switch(Op->OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
}
}
switch (OpSize) {
case 1: {
reinterpret_cast<std::atomic<uint8_t>*>(MemData)->store(*GetSrc<uint8_t*>(Data->SSAData, Op->Value));
break;
}
case 2: {
reinterpret_cast<std::atomic<uint16_t>*>(MemData)->store(*GetSrc<uint16_t*>(Data->SSAData, Op->Value));
break;
}
case 4: {
reinterpret_cast<std::atomic<uint32_t>*>(MemData)->store(*GetSrc<uint32_t*>(Data->SSAData, Op->Value));
break;
}
case 8: {
reinterpret_cast<std::atomic<uint64_t>*>(MemData)->store(*GetSrc<uint64_t*>(Data->SSAData, Op->Value));
break;
}
default:
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
break;
}
}
DEF_OP(VLoadMemElement) {
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Value);
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Addr), 16);
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
MemData, Op->Header.ElementSize);
}
DEF_OP(VStoreMemElement) {
#define STORE_DATA(x, y) \
case x: { \
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Value); \
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Addr)[Op->Index], sizeof(y)); \
break; \
}
auto Op = IROp->C<IR::IROp_VStoreMemElement>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
STORE_DATA(1, uint8_t)
STORE_DATA(2, uint16_t)
STORE_DATA(4, uint32_t)
STORE_DATA(8, uint64_t)
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size"); break;
}
#undef STORE_DATA
}
DEF_OP(CacheLineClear) {
auto Op = IROp->C<IR::IROp_CacheLineClear>();
char *MemData = *GetSrc<char **>(Data->SSAData, Op->Addr);
// 64-byte cache line clear
CacheLineFlush(MemData);
}
DEF_OP(CacheLineZero) {
auto Op = IROp->C<IR::IROp_CacheLineZero>();
uintptr_t MemData = *GetSrc<uintptr_t*>(Data->SSAData, Op->Addr);
// Force cacheline alignment
MemData = MemData & ~(CPUIDEmu::CACHELINE_SIZE - 1);
using DataType = uint64_t;
DataType *MemData64 = reinterpret_cast<DataType*>(MemData);
// 64-byte cache line zero
for (size_t i = 0; i < (CPUIDEmu::CACHELINE_SIZE / sizeof(DataType)); ++i) {
MemData64[i] = 0;
}
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,174 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <FEXHeaderUtils/Syscalls.h>
#include <cstdint>
#ifdef _M_X86_64
#include <xmmintrin.h>
#endif
#include <sys/random.h>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(Fence) {
auto Op = IROp->C<IR::IROp_Fence>();
switch (Op->Fence) {
case IR::Fence_Load.Val:
std::atomic_thread_fence(std::memory_order_acquire);
break;
case IR::Fence_LoadStore.Val:
std::atomic_thread_fence(std::memory_order_seq_cst);
break;
case IR::Fence_Store.Val:
std::atomic_thread_fence(std::memory_order_release);
break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
}
}
DEF_OP(Break) {
auto Op = IROp->C<IR::IROp_Break>();
Data->State->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = 1;
Data->State->CurrentFrame->SynchronousFaultData.Signal = Op->Reason.Signal;
Data->State->CurrentFrame->SynchronousFaultData.TrapNo = Op->Reason.TrapNumber;
Data->State->CurrentFrame->SynchronousFaultData.err_code = Op->Reason.ErrorRegister;
Data->State->CurrentFrame->SynchronousFaultData.si_code = Op->Reason.si_code;
switch (Op->Reason.Signal) {
case SIGILL:
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
break;
case SIGTRAP:
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
break;
case SIGSEGV:
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGSEGV);
break;
default:
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
break;
}
}
DEF_OP(GetRoundingMode) {
uint32_t GuestRounding{};
#ifdef _M_ARM_64
uint64_t Tmp{};
__asm(R"(
mrs %[Tmp], FPCR;
)"
: [Tmp] "=r" (Tmp));
// Extract the rounding
// On ARM the ordering is different than on x86
GuestRounding |= ((Tmp >> 24) & 1) ? IR::ROUND_MODE_FLUSH_TO_ZERO : 0;
uint8_t RoundingMode = (Tmp >> 22) & 0b11;
if (RoundingMode == 0)
GuestRounding |= IR::ROUND_MODE_NEAREST;
else if (RoundingMode == 1)
GuestRounding |= IR::ROUND_MODE_POSITIVE_INFINITY;
else if (RoundingMode == 2)
GuestRounding |= IR::ROUND_MODE_NEGATIVE_INFINITY;
else if (RoundingMode == 3)
GuestRounding |= IR::ROUND_MODE_TOWARDS_ZERO;
#else
GuestRounding = _mm_getcsr();
// Extract the rounding
GuestRounding = (GuestRounding >> 13) & 0b111;
#endif
memcpy(GDP, &GuestRounding, sizeof(GuestRounding));
}
DEF_OP(SetRoundingMode) {
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
const auto GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->RoundMode);
#ifdef _M_ARM_64
uint64_t HostRounding{};
__asm volatile(R"(
mrs %[Tmp], FPCR;
)"
: [Tmp] "=r" (HostRounding));
// Mask out the rounding
HostRounding &= ~(0b111 << 22);
HostRounding |= (GuestRounding & IR::ROUND_MODE_FLUSH_TO_ZERO) ? (1U << 24) : 0;
uint8_t RoundingMode = GuestRounding & 0b11;
if (RoundingMode == IR::ROUND_MODE_NEAREST)
HostRounding |= (0b00U << 22);
else if (RoundingMode == IR::ROUND_MODE_POSITIVE_INFINITY)
HostRounding |= (0b01U << 22);
else if (RoundingMode == IR::ROUND_MODE_NEGATIVE_INFINITY)
HostRounding |= (0b10U << 22);
else if (RoundingMode == IR::ROUND_MODE_TOWARDS_ZERO)
HostRounding |= (0b11U << 22);
__asm volatile(R"(
msr FPCR, %[Tmp];
)"
:: [Tmp] "r" (HostRounding));
#else
uint32_t HostRounding = _mm_getcsr();
// Cut out the host rounding mode
HostRounding &= ~(0b111 << 13);
// Insert our new rounding mode
HostRounding |= GuestRounding << 13;
_mm_setcsr(HostRounding);
#endif
}
DEF_OP(Print) {
auto Op = IROp->C<IR::IROp_Print>();
const uint8_t OpSize = IROp->Size;
if (OpSize <= 8) {
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
LogMan::Msg::IFmt(">>>> Value in Arg: 0x{:x}, {}", Src, Src);
}
else if (OpSize == 16) {
const auto Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Value);
const uint64_t Src0 = Src;
const uint64_t Src1 = Src >> 64;
LogMan::Msg::IFmt(">>>> Value[0] in Arg: 0x{:x}, {}", Src0, Src0);
LogMan::Msg::IFmt(" Value[1] in Arg: 0x{:x}, {}", Src1, Src1);
}
else
LOGMAN_MSG_A_FMT("Unknown value size: {}", OpSize);
}
DEF_OP(ProcessorID) {
uint32_t CPU, CPUNode;
FHU::Syscalls::getcpu(&CPU, &CPUNode);
GD = (CPUNode << 12) | CPU;
}
DEF_OP(RDRAND) {
// We are ignoring Op->GetReseeded in the interpreter
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
ssize_t Result = ::getrandom(&DstPtr[0], 8, 0);
// Second result is if we managed to read a valid random number or not
DstPtr[1] = Result == 8 ? 1 : 0;
}
DEF_OP(Yield) {
// Nop implementation
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,35 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(ExtractElementPair) {
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
const auto Src = GetSrc<uintptr_t>(Data->SSAData, Op->Pair);
memcpy(GDP,
reinterpret_cast<void*>(Src + Op->Header.Size * Op->Element), Op->Header.Size);
}
DEF_OP(CreateElementPair) {
auto Op = IROp->C<IR::IROp_CreateElementPair>();
const void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Lower);
const void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Upper);
uint8_t *Dst = GetDest<uint8_t*>(Data->SSAData, Node);
memcpy(Dst, Src_Lower, IROp->ElementSize);
memcpy(Dst + IROp->ElementSize, Src_Upper, IROp->ElementSize);
}
#undef DEF_OP
} // namespace FEXCore::CPU
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -1,131 +0,0 @@
/*
$info$
tags: backend|arm64
desc: relocation logic of the arm64 splatter backend
$end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "Interface/HLE/Thunks/Thunks.h"
namespace FEXCore::CPU {
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
switch (Op) {
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
break;
default:
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
break;
}
return ~0ULL;
}
void Arm64JITCore::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
Relocation MoveABI{};
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t *>();
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
MoveABI.NamedThunkMove.Symbol = Sum;
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
uint64_t Pointer = GetNamedSymbolLiteral(Op);
Arm64JITCore::NamedSymbolLiteralPair Lit {
.Lit = Literal(Pointer),
.MoveABI = {
.NamedSymbolLiteral = {
.Header = {
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
},
.Symbol = Op,
.Offset = 0,
},
},
};
return Lit;
}
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t *>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
place(&Lit.Lit);
Relocations.emplace_back(Lit.MoveABI);
}
void Arm64JITCore::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
Relocation MoveABI{};
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t *>();
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
MoveABI.GuestRIPMove.GuestRIP = Constant;
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
size_t DataIndex{};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
Literal<uint64_t> Lit(Pointer);
place(&Lit);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
// XXX: Reenable once the JIT Object Cache is upstream
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
}
}
return true;
}
}
+165 -214
View File
@@ -4,26 +4,26 @@ tags: backend|arm64
$end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CASPair>();
uint8_t OpSize = IROp->Size;
// Size is the size of each pair element
auto Dst = GetSrcPair<RA_64>(Node);
auto Expected = GetSrcPair<RA_64>(Op->Expected.ID());
auto Desired = GetSrcPair<RA_64>(Op->Desired.ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto Expected = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
auto Desired = GetSrcPair<RA_64>(Op->Header.Args[1].ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
if (CTX->HostFeatures.SupportsAtomics) {
if (SupportsAtomics) {
mov(TMP3, Expected.first);
mov(TMP4, Expected.second);
switch (IROp->ElementSize) {
switch (OpSize) {
case 4:
caspal(TMP3.W(), TMP4.W(), Desired.first.W(), Desired.second.W(), MemOperand(MemSrc));
mov(Dst.first.W(), TMP3.W());
@@ -34,17 +34,16 @@ DEF_OP(CASPair) {
mov(Dst.first, TMP3);
mov(Dst.second, TMP4);
break;
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
default: LogMan::Msg::A("Unsupported: %d", OpSize);
}
}
else {
switch (IROp->ElementSize) {
switch (OpSize) {
case 4: {
aarch64::Label LoopTop;
aarch64::Label LoopNotExpected;
aarch64::Label LoopExpected;
bind(&LoopTop);
ldaxp(TMP2.W(), TMP3.W(), MemOperand(MemSrc));
cmp(TMP2.W(), Expected.first.W());
ccmp(TMP3.W(), Expected.second.W(), NoFlag, Condition::eq);
@@ -70,7 +69,6 @@ DEF_OP(CASPair) {
aarch64::Label LoopNotExpected;
aarch64::Label LoopExpected;
bind(&LoopTop);
ldaxp(TMP2.X(), TMP3.X(), MemOperand(MemSrc));
cmp(TMP2.X(), Expected.first.X());
ccmp(TMP3.X(), Expected.second.X(), NoFlag, Condition::eq);
@@ -91,7 +89,7 @@ DEF_OP(CASPair) {
bind(&LoopExpected);
break;
}
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
default: LogMan::Msg::A("Unsupported: %d", OpSize);
}
}
}
@@ -99,22 +97,25 @@ DEF_OP(CASPair) {
DEF_OP(CAS) {
auto Op = IROp->C<IR::IROp_CAS>();
uint8_t OpSize = IROp->Size;
// Args[0]: Expected
// Args[1]: Desired
// Args[2]: Pointer
// DataSrc = *Src1
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
// This will write to memory! Careful!
auto Expected = GetReg<RA_64>(Op->Expected.ID());
auto Desired = GetReg<RA_64>(Op->Desired.ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto Expected = GetReg<RA_64>(Op->Header.Args[0].ID());
auto Desired = GetReg<RA_64>(Op->Header.Args[1].ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
if (CTX->HostFeatures.SupportsAtomics) {
if (SupportsAtomics) {
mov(TMP2, Expected);
switch (OpSize) {
case 1: casalb(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
case 2: casalh(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
case 4: casal(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
case 8: casal(TMP2.X(), Desired.X(), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
default: LogMan::Msg::A("Unsupported: %d", OpSize);
}
mov(GetReg<RA_64>(Node), TMP2);
}
@@ -205,7 +206,7 @@ DEF_OP(CAS) {
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", OpSize);
default: LogMan::Msg::A("Unhandled Atomic size: %d", OpSize);
}
}
}
@@ -213,25 +214,25 @@ DEF_OP(CAS) {
DEF_OP(AtomicAdd) {
auto Op = IROp->C<IR::IROp_AtomicAdd>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: staddlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 2: staddlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 4: staddl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 8: staddl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
if (SupportsAtomics) {
switch (Op->Size) {
case 1: staddlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 2: staddlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 4: staddl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 8: staddl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -240,7 +241,7 @@ DEF_OP(AtomicAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -249,7 +250,7 @@ DEF_OP(AtomicAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -258,12 +259,12 @@ DEF_OP(AtomicAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
add(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
add(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -271,26 +272,26 @@ DEF_OP(AtomicAdd) {
DEF_OP(AtomicSub) {
auto Op = IROp->C<IR::IROp_AtomicSub>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
neg(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
if (SupportsAtomics) {
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (Op->Size) {
case 1: staddlb(TMP2.W(), MemOperand(MemSrc)); break;
case 2: staddlh(TMP2.W(), MemOperand(MemSrc)); break;
case 4: staddl(TMP2.W(), MemOperand(MemSrc)); break;
case 8: staddl(TMP2.X(), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -299,7 +300,7 @@ DEF_OP(AtomicSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -308,7 +309,7 @@ DEF_OP(AtomicSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -317,12 +318,12 @@ DEF_OP(AtomicSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
sub(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
sub(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -330,26 +331,26 @@ DEF_OP(AtomicSub) {
DEF_OP(AtomicAnd) {
auto Op = IROp->C<IR::IROp_AtomicAnd>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
mvn(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
if (SupportsAtomics) {
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (Op->Size) {
case 1: stclrlb(TMP2.W(), MemOperand(MemSrc)); break;
case 2: stclrlh(TMP2.W(), MemOperand(MemSrc)); break;
case 4: stclrl(TMP2.W(), MemOperand(MemSrc)); break;
case 8: stclrl(TMP2.X(), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -358,7 +359,7 @@ DEF_OP(AtomicAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -367,7 +368,7 @@ DEF_OP(AtomicAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -376,12 +377,12 @@ DEF_OP(AtomicAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
and_(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
and_(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -389,25 +390,25 @@ DEF_OP(AtomicAnd) {
DEF_OP(AtomicOr) {
auto Op = IROp->C<IR::IROp_AtomicOr>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: stsetlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 2: stsetlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 4: stsetl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 8: stsetl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
if (SupportsAtomics) {
switch (Op->Size) {
case 1: stsetlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 2: stsetlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 4: stsetl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 8: stsetl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -416,7 +417,7 @@ DEF_OP(AtomicOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -425,7 +426,7 @@ DEF_OP(AtomicOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -434,12 +435,12 @@ DEF_OP(AtomicOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
orr(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
orr(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -447,25 +448,25 @@ DEF_OP(AtomicOr) {
DEF_OP(AtomicXor) {
auto Op = IROp->C<IR::IROp_AtomicXor>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: steorlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 2: steorlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 4: steorl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 8: steorl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
if (SupportsAtomics) {
switch (Op->Size) {
case 1: steorlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 2: steorlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 4: steorl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 8: steorl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -474,7 +475,7 @@ DEF_OP(AtomicXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -483,7 +484,7 @@ DEF_OP(AtomicXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -492,12 +493,12 @@ DEF_OP(AtomicXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
eor(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
eor(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -505,44 +506,45 @@ DEF_OP(AtomicXor) {
DEF_OP(AtomicSwap) {
auto Op = IROp->C<IR::IROp_AtomicSwap>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
mov(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
case 1: swpalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: swpalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: swpal(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: swpal(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
if (SupportsAtomics) {
mov(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (Op->Size) {
case 1: swplb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: swplh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: swpl(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: swpl(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
mov(TMP3, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
stlxrb(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
uxtb(GetReg<RA_32>(Node), TMP2.W());
uxtb(GetReg<RA_64>(Node), TMP2.W());
break;
}
case 2: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
stlxrh(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
uxtw(GetReg<RA_32>(Node), TMP2.W());
uxtw(GetReg<RA_64>(Node), TMP2.W());
break;
}
case 4: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
stlxr(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
break;
@@ -551,37 +553,37 @@ DEF_OP(AtomicSwap) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
stlxr(TMP4, GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc));
stlxr(TMP4, TMP3.X(), MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2.X());
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
DEF_OP(AtomicFetchAdd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: ldaddalb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldaddalh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldaddal(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldaddal(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
if (SupportsAtomics) {
switch (Op->Size) {
case 1: ldaddalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldaddalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldaddal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldaddal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -591,7 +593,7 @@ DEF_OP(AtomicFetchAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -601,7 +603,7 @@ DEF_OP(AtomicFetchAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -611,39 +613,39 @@ DEF_OP(AtomicFetchAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
add(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
add(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
DEF_OP(AtomicFetchSub) {
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
neg(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
if (SupportsAtomics) {
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (Op->Size) {
case 1: ldaddalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldaddalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldaddal(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldaddal(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -653,7 +655,7 @@ DEF_OP(AtomicFetchSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -663,7 +665,7 @@ DEF_OP(AtomicFetchSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -673,39 +675,39 @@ DEF_OP(AtomicFetchSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
sub(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
sub(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
DEF_OP(AtomicFetchAnd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
mvn(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
if (SupportsAtomics) {
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (Op->Size) {
case 1: ldclralb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldclralh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldclral(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldclral(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -715,7 +717,7 @@ DEF_OP(AtomicFetchAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -725,7 +727,7 @@ DEF_OP(AtomicFetchAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -735,38 +737,38 @@ DEF_OP(AtomicFetchAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
and_(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
and_(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: ldsetalb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldsetalh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldsetal(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldsetal(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
if (SupportsAtomics) {
switch (Op->Size) {
case 1: ldsetalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldsetalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldsetal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldsetal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -776,7 +778,7 @@ DEF_OP(AtomicFetchOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -786,7 +788,7 @@ DEF_OP(AtomicFetchOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -796,38 +798,38 @@ DEF_OP(AtomicFetchOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
orr(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
orr(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
DEF_OP(AtomicFetchXor) {
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: ldeoralb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldeoralh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldeoral(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldeoral(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
if (SupportsAtomics) {
switch (Op->Size) {
case 1: ldeoralb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldeoralh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldeoral(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldeoral(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -837,7 +839,7 @@ DEF_OP(AtomicFetchXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -847,7 +849,7 @@ DEF_OP(AtomicFetchXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -857,67 +859,17 @@ DEF_OP(AtomicFetchXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
eor(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
eor(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
DEF_OP(AtomicFetchNeg) {
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
// TMP2-TMP3
switch (IROp->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
neg(TMP3.W(), TMP2.W());
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
break;
}
case 2: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
neg(TMP3.W(), TMP2.W());
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
break;
}
case 4: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
neg(TMP3.W(), TMP2.W());
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
break;
}
case 8: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
neg(TMP3, TMP2);
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
#undef DEF_OP
void Arm64JITCore::RegisterAtomicHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
@@ -934,7 +886,6 @@ void Arm64JITCore::RegisterAtomicHandlers() {
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
#undef REGISTER_OP
}
}
+98 -235
View File
@@ -4,22 +4,28 @@ tags: backend|arm64
$end_info$
*/
#include "Interface/Context/Context.h"
#include "FEXCore/IR/IR.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "Interface/Core/InternalThreadState.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/MathUtils.h>
#include <Interface/HLE/Thunks/Thunks.h>
namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(GuestCallDirect) {
LogMan::Msg::D("Unimplemented");
}
DEF_OP(GuestCallIndirect) {
LogMan::Msg::D("Unimplemented");
}
DEF_OP(GuestReturn) {
LogMan::Msg::D("Unimplemented");
}
DEF_OP(SignalReturn) {
// First we must reset the stack
@@ -27,7 +33,7 @@ DEF_OP(SignalReturn) {
// Now branch to our signal return helper
// This can't be a direct branch since the code needs to live at a constant location
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)));
LoadConstant(x0, ThreadSharedData.SignalReturnInstruction);
br(x0);
}
@@ -40,10 +46,10 @@ DEF_OP(CallbackReturn) {
ResetStack();
// We can now lower the ref counter again
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
LoadConstant(x0, reinterpret_cast<uint64_t>(ThreadSharedData.SignalHandlerRefCounterPtr));
ldr(w2, MemOperand(x0));
sub(w2, w2, 1);
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
str(w2, MemOperand(x0));
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
@@ -67,7 +73,7 @@ DEF_OP(ExitFunction) {
uint64_t NewRIP;
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
Literal l_BranchHost{ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker};
Literal l_BranchHost{Dispatcher->ExitFunctionLinkerAddress};
Literal l_BranchGuest{NewRIP};
ldr(x0, &l_BranchHost);
@@ -76,10 +82,10 @@ DEF_OP(ExitFunction) {
place(&l_BranchHost);
place(&l_BranchGuest);
} else {
RipReg = GetReg<RA_64>(Op->NewRIP.ID());
RipReg = GetReg<RA_64>(Op->Header.Args[0].ID());
// L1 Cache
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)));
LoadConstant(x0, ThreadState->LookupCache->GetL1Pointer());
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x3, Shift::LSL, 4));
@@ -90,23 +96,30 @@ DEF_OP(ExitFunction) {
br(x1);
bind(&FullLookup);
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop)));
LoadConstant(TMP1, Dispatcher->AbsoluteLoopTopAddress);
str(RipReg, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
br(TMP1);
}
}
DEF_OP(Jump) {
const auto Op = IROp->C<IR::IROp_Jump>();
const auto Target = Op->TargetBlock.ID();
auto Op = IROp->C<IR::IROp_Jump>();
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
Label *TargetLabel;
auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID());
if (IsTarget == JumpTargets.end()) {
TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second;
}
else {
TargetLabel = &IsTarget->second;
}
PendingTargetLabel = TargetLabel;
}
#define GRCMP(Node) (Op->CompareSize == 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
#define GRFCMP(Node) (Op->CompareSize == 4 ? GetDst(Node).S() : GetDst(Node).D())
static Condition MapBranchCC(IR::CondClassType Cond) {
Condition MapBranchCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_EQ: return Condition::eq;
case FEXCore::IR::COND_NEQ: return Condition::ne;
@@ -121,7 +134,7 @@ static Condition MapBranchCC(IR::CondClassType Cond) {
case FEXCore::IR::COND_FLU: return Condition::lt;
case FEXCore::IR::COND_FGE: return Condition::ge;
case FEXCore::IR::COND_FLEU:return Condition::le;
case FEXCore::IR::COND_FGT: return Condition::gt;
case FEXCore::IR::COND_FGT: return Condition::hi;
case FEXCore::IR::COND_FU: return Condition::vs;
case FEXCore::IR::COND_FNU: return Condition::vc;
case FEXCore::IR::COND_VS:
@@ -129,7 +142,7 @@ static Condition MapBranchCC(IR::CondClassType Cond) {
case FEXCore::IR::COND_MI:
case FEXCore::IR::COND_PL:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
LogMan::Msg::A("Unsupported compare type");
return Condition::nv;
}
}
@@ -138,34 +151,51 @@ static Condition MapBranchCC(IR::CondClassType Cond) {
DEF_OP(CondJump) {
auto Op = IROp->C<IR::IROp_CondJump>();
Label *TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
Label *TrueTargetLabel;
Label *FalseTargetLabel;
auto TrueIter = JumpTargets.find(Op->TrueBlock.ID());
auto FalseIter = JumpTargets.find(Op->FalseBlock.ID());
if (TrueIter == JumpTargets.end()) {
TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
}
else {
TrueTargetLabel = &TrueIter->second;
}
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
bool isConst = IsInlineConstant(Op->Cmp2, &Const);
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
LogMan::Throw::A(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
cbz(GRCMP(Op->Cmp1.ID()), TrueTargetLabel);
} else if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_NEQ) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
LogMan::Throw::A(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
cbnz(GRCMP(Op->Cmp1.ID()), TrueTargetLabel);
} else {
if (IsGPR(Op->Cmp1.ID())) {
if (isConst) {
if (isConst)
cmp(GRCMP(Op->Cmp1.ID()), Const);
} else {
else
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
}
} else if (IsFPR(Op->Cmp1.ID())) {
fcmp(GRFCMP(Op->Cmp1.ID()), GRFCMP(Op->Cmp2.ID()));
} else {
LOGMAN_MSG_A_FMT("CondJump: Expected GPR or FPR");
LogMan::Msg::A("CondJump: Expected GPR or FPR");
}
b(TrueTargetLabel, MapBranchCC(Op->Cond));
}
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
if (FalseIter == JumpTargets.end()) {
FalseTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
}
else {
FalseTargetLabel = &FalseIter->second;
}
PendingTargetLabel = FalseTargetLabel;
}
DEF_OP(Syscall) {
@@ -175,16 +205,8 @@ DEF_OP(Syscall) {
// X1: ThreadState
// X2: Pointer to SyscallArguments
FEXCore::IR::SyscallFlags Flags = Op->Flags;
PushDynamicRegsAndLR();
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
SpillStaticRegs();
}
else {
// Need to spill all caller saved registers still
SpillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
}
SpillStaticRegs();
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
sub(sp, sp, SPOffset);
@@ -193,181 +215,22 @@ DEF_OP(Syscall) {
str(GetReg<RA_64>(Op->Header.Args[i].ID()), MemOperand(sp, i * 8));
}
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj)));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)));
LoadConstant(x0, reinterpret_cast<uint64_t>(CTX->SyscallHandler));
mov(x1, STATE);
mov(x2, sp);
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(x3);
#else
LoadConstant(x3, reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall));
blr(x3);
#endif
add(sp, sp, SPOffset);
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY &&
(Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
FillStaticRegs();
}
else {
// Result is now in x0
// Fix the stack and any values that were stepped on
FillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
}
// Result is now in x0
// Fix the stack and any values that were stepped on
FillStaticRegs();
PopDynamicRegsAndLR();
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
// Move result to its destination register
mov(GetReg<RA_64>(Node), x0);
}
}
DEF_OP(InlineSyscall) {
auto Op = IROp->C<IR::IROp_InlineSyscall>();
// Arguments are passed as follows:
// X8: SyscallNumber - RA INTERSECT
// X0: Arg0 & Return
// X1: Arg1
// X2: Arg2
// X3: Arg3
// X4: Arg4 - RA INTERSECT
// X5: Arg5 - RA INTERSECT
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
const static std::array<vixl::aarch64::Register, FEXCore::HLE::SyscallArguments::MAX_ARGS-1> RegArgs = {{
x0, x1, x2, x3, x4, x5
}};
bool Intersects{};
// We always need to spill x8 since we can't know if it is live at this SSA location
uint32_t SpillMask = 1U << 8;
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
if (Reg.GetCode() == x8.GetCode() ||
Reg.GetCode() == x4.GetCode() ||
Reg.GetCode() == x5.GetCode()) {
SpillMask |= (1U << Reg.GetCode());
Intersects = true;
}
}
// XXX: For some reason spilling only the x4, x5, and x8 registers was causing issues
// Come back to this once investigation reveals why it fails the gvisor ioctl test
// For now override to all GPRs
SpillMask = ~0U;
// Ordering is incredibly important here
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
// Only spill the registers that intersect with our usage
SpillStaticRegs(false, SpillMask);
// Now that we are spilled, store in the state that we are in a syscall
// Still without overwriting registers that matter
// 16bit LoadConstant to be a single instruction
// We must always spill at least one register (x8) so this value always has a bit set
// This gives the signal handler a value to check to see if we are in a syscall at all
LoadConstant(x0, SpillMask & 0xFFFF);
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
// Now that we have claimed to be a syscall we can set up the arguments
if (Intersects) {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
if (CTX->Config.Is64BitMode()) {
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
// In the case of intersection with x4, x5, or x8 then these are currently SRA
// for registers RAX, RBX, and RSI. Which have just been spilled
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
if (Reg.GetCode() == x8.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
}
else if (Reg.GetCode() == x4.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
}
else if (Reg.GetCode() == x5.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
}
}
}
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
if (CTX->Config.Is64BitMode()) {
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
// In the case of intersection with x4, x5, or x8 then these are currently SRA
// for registers RAX, RBX, and RSI. Which have just been spilled
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
if (Reg.GetCode() == x8.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
}
else if (Reg.GetCode() == x4.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
}
else if (Reg.GetCode() == x5.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
}
else {
mov(RegArgs[i], Reg);
}
}
else {
auto Reg = GetReg<RA_32>(Op->Header.Args[i].ID());
if (Reg.GetCode() == w8.GetCode()) {
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
}
else if (Reg.GetCode() == w4.GetCode()) {
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
}
else if (Reg.GetCode() == w5.GetCode()) {
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
}
else {
uxtw(RegArgs[i].W(), Reg);
}
}
}
}
else {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
if (CTX->Config.Is64BitMode()) {
mov(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
}
else {
uxtw(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
}
}
}
LoadConstant(x8, Op->HostSyscallNumber);
svc(0);
// On updated signal mask we can receive a signal RIGHT HERE
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
// Now that we are done in the syscall we need to carefully peel back the state
// First unspill the registers from before
FillStaticRegs(false, SpillMask);
// Now the registers we've spilled are back in their original host registers
// We can safely claim we are no longer in a syscall
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
// Result is now in x0
// Move result to its destination register
if (CTX->Config.Is64BitMode()) {
mov(GetReg<RA_64>(Node), x0);
}
else {
uxtw(GetReg<RA_64>(Node), x0);
}
}
// Move result to its destination register
mov(GetReg<RA_64>(Node), x0);
}
DEF_OP(Thunk) {
@@ -380,35 +243,32 @@ DEF_OP(Thunk) {
PushDynamicRegsAndLR();
mov(x0, GetReg<RA_64>(Op->ArgPtr.ID()));
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
LoadConstant(x2, (uintptr_t)thunkFn);
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<void, void*, void*>(x2);
#else
blr(x2);
#endif
PopDynamicRegsAndLR();
FillStaticRegs(); // load from ctx after ra64 refill
}
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
const auto *OldCode = (const uint8_t *)&Op->CodeOriginalLow;
uint8_t *OldCode = (uint8_t *)&Op->CodeOriginalLow;
int len = Op->CodeLength;
int idx = 0;
LoadConstant(GetReg<RA_64>(Node), 0);
LoadConstant(x0, Entry + Op->Offset);
LoadConstant(x0, IR->GetHeader()->Entry + Op->Offset);
LoadConstant(x1, 1);
while (len >= 8)
{
ldr(x2, MemOperand(x0, idx));
LoadConstant(x3, *(const uint32_t *)(OldCode + idx));
LoadConstant(x3, *(uint32_t *)(OldCode + idx));
cmp(x2, x3);
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
len -= 8;
@@ -417,7 +277,7 @@ DEF_OP(ValidateCode) {
while (len >= 4)
{
ldr(w2, MemOperand(x0, idx));
LoadConstant(w3, *(const uint32_t *)(OldCode + idx));
LoadConstant(w3, *(uint32_t *)(OldCode + idx));
cmp(w2, w3);
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
len -= 4;
@@ -426,7 +286,7 @@ DEF_OP(ValidateCode) {
while (len >= 2)
{
ldrh(w2, MemOperand(x0, idx));
LoadConstant(w3, *(const uint16_t *)(OldCode + idx));
LoadConstant(w3, *(uint16_t *)(OldCode + idx));
cmp(w2, w3);
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
len -= 2;
@@ -435,7 +295,7 @@ DEF_OP(ValidateCode) {
while (len >= 1)
{
ldrb(w2, MemOperand(x0, idx));
LoadConstant(w3, *(const uint8_t *)(OldCode + idx));
LoadConstant(w3, *(uint8_t *)(OldCode + idx));
cmp(w2, w3);
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
len -= 1;
@@ -443,7 +303,7 @@ DEF_OP(ValidateCode) {
}
}
DEF_OP(ThreadRemoveCodeEntry) {
DEF_OP(RemoveCodeEntry) {
// Arguments are passed as follows:
// X0: Thread
// X1: RIP
@@ -451,15 +311,11 @@ DEF_OP(ThreadRemoveCodeEntry) {
PushDynamicRegsAndLR();
mov(x0, STATE);
LoadConstant(x1, Entry);
LoadConstant(x1, IR->GetHeader()->Entry);
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)));
LoadConstant(x2, reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit));
SpillStaticRegs();
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<void, void*, void*>(x2);
#else
blr(x2);
#endif
FillStaticRegs();
// Fix the stack and any values that were stepped on
@@ -470,22 +326,27 @@ DEF_OP(CPUID) {
auto Op = IROp->C<IR::IROp_CPUID>();
PushDynamicRegsAndLR();
SpillStaticRegs();
// x0 = CPUID Handler
// x1 = CPUID Function
// x2 = CPUID Leaf
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)));
mov(x1, GetReg<RA_64>(Op->Function.ID()));
mov(x2, GetReg<RA_64>(Op->Leaf.ID()));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<__uint128_t, void*, uint64_t, uint64_t>(x3);
#else
blr(x3);
#endif
LoadConstant(x0, reinterpret_cast<uint64_t>(&CTX->CPUID));
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
mov(x2, GetReg<RA_64>(Op->Header.Args[1].ID()));
using ClassPtrType = FEXCore::CPUID::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t, uint32_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
PtrCast Ptr;
Ptr.ClassPtr = &FEXCore::CPUIDEmu::RunFunction;
LoadConstant(x3, Ptr.Data);
SpillStaticRegs();
blr(x3);
FillStaticRegs();
PopDynamicRegsAndLR();
// Results are in x0, x1
@@ -498,16 +359,18 @@ DEF_OP(CPUID) {
#undef DEF_OP
void Arm64JITCore::RegisterBranchHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
REGISTER_OP(GUESTRETURN, GuestReturn);
REGISTER_OP(SIGNALRETURN, SignalReturn);
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
REGISTER_OP(EXITFUNCTION, ExitFunction);
REGISTER_OP(JUMP, Jump);
REGISTER_OP(CONDJUMP, CondJump);
REGISTER_OP(SYSCALL, Syscall);
REGISTER_OP(INLINESYSCALL, InlineSyscall);
REGISTER_OP(THUNK, Thunk);
REGISTER_OP(VALIDATECODE, ValidateCode);
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
REGISTER_OP(CPUID, CPUID);
#undef REGISTER_OP
}
@@ -10,117 +10,28 @@ namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(VInsGPR) {
const auto Op = IROp->C<IR::IROp_VInsGPR>();
const auto OpSize = IROp->Size;
const auto DestIdx = Op->DestIdx;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto DestVector = GetSrc(Op->DestVector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto ElementSizeBits = ElementSize * 8;
const auto Offset = ElementSizeBits * DestIdx;
const auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
const auto InUpperLane = Offset >= SSEBitSize;
// This is going to be a little gross. Pls forgive me.
// Since SVE has the whole vector length agnostic programming
// thing going on, we can't exactly freely insert entries into
// arbitrary locations in the vector.
//
// SVE *does* have INSR, however this only shifts the entire
// vector to the left by an element size and inserts a value
// at the beginning of the vector. Not *quite* what we need.
// (though INSR *is* very useful for other things).
//
// The idea is (in the case of the upper lane), move the upper
// lane down, insert into it and recombine with the lower lane.
//
// In the case of the lower lane, insert and then recombine with
// the upper lane.
if (InUpperLane) {
// Move the upper lane down for the insertion.
const auto CompactPred = p0;
not_(CompactPred.VnB(), PRED_TMP_32B.Zeroing(), PRED_TMP_16B.VnB());
compact(VTMP1.Z().VnD(), CompactPred, DestVector.Z().VnD());
auto Op = IROp->C<IR::IROp_VInsGPR>();
mov(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
switch (Op->Header.ElementSize) {
case 1: {
ins(GetDst(Node).V16B(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
break;
}
// Put data in place for destructive SPLICE below.
mov(Dst.Z().VnD(), DestVector.Z().VnD());
// Inserts the GPR value into the given V register.
// Also automatically adjusts the index in the case of using the
// moved upper lane.
const auto Insert = [&](const aarch64::VRegister& reg, int index) {
switch (ElementSize) {
case 1:
if (InUpperLane) {
index -= 16;
}
ins(reg.V16B(), index, GetReg<RA_32>(Op->Src.ID()));
break;
case 2:
if (InUpperLane) {
index -= 8;
}
ins(reg.V8H(), index, GetReg<RA_32>(Op->Src.ID()));
break;
case 4:
if (InUpperLane) {
index -= 4;
}
ins(reg.V4S(), index, GetReg<RA_32>(Op->Src.ID()));
break;
case 8:
if (InUpperLane) {
index -= 2;
}
ins(reg.V2D(), index, GetReg<RA_64>(Op->Src.ID()));
break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
};
if (InUpperLane) {
Insert(VTMP1, DestIdx);
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), VTMP1.Z().VnD());
} else {
Insert(Dst, DestIdx);
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), DestVector.Z().VnD());
case 2: {
ins(GetDst(Node).V8H(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
break;
}
} else {
mov(Dst, DestVector);
switch (ElementSize) {
case 1: {
ins(Dst.V16B(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
break;
}
case 2: {
ins(Dst.V8H(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
break;
}
case 4: {
ins(Dst.V4S(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
break;
}
case 8: {
ins(Dst.V2D(), DestIdx, GetReg<RA_64>(Op->Src.ID()));
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
case 4: {
ins(GetDst(Node).V4S(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
break;
}
case 8: {
ins(GetDst(Node).V2D(), Op->Index, GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
}
default: LogMan::Msg::A("Unknown Element Size: %d", Op->Header.ElementSize); break;
}
}
@@ -128,443 +39,162 @@ DEF_OP(VCastFromGPR) {
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
switch (Op->Header.ElementSize) {
case 1:
uxtb(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
uxtb(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
fmov(GetDst(Node).S(), TMP1.W());
break;
case 2:
uxth(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
uxth(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
fmov(GetDst(Node).S(), TMP1.W());
break;
case 4:
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()).W());
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
break;
case 8:
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()).X());
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()).X());
break;
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Float_FromGPR_U) {
LogMan::Msg::A("Unimplemented");
}
DEF_OP(Float_FromGPR_S) {
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
const uint16_t ElementSize = Op->Header.ElementSize;
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0404: { // Float <- int32_t
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()));
break;
}
case 0x0408: { // Float <- int64_t
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Src.ID()));
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Header.Args[0].ID()));
break;
}
case 0x0804: { // Double <- int32_t
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Src.ID()));
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Header.Args[0].ID()));
break;
}
case 0x0808: { // Double <- int64_t
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()));
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
Conv, ElementSize, Op->SrcElementSize);
break;
}
}
DEF_OP(Float_FToF) {
auto Op = IROp->C<IR::IROp_Float_FToF>();
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0804: { // Double <- Float
fcvt(GetDst(Node).D(), GetSrc(Op->Scalar.ID()).S());
fcvt(GetDst(Node).D(), GetSrc(Op->Header.Args[0].ID()).S());
break;
}
case 0x0408: { // Float <- Double
fcvt(GetDst(Node).S(), GetSrc(Op->Scalar.ID()).D());
fcvt(GetDst(Node).S(), GetSrc(Op->Header.Args[0].ID()).D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
default: LogMan::Msg::A("Unknown FCVT sizes: 0x%x", Conv);
}
}
DEF_OP(Vector_UToF) {
auto Op = IROp->C<IR::IROp_Vector_UToF>();
switch (Op->Header.ElementSize) {
case 4:
ucvtf(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
ucvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Vector_SToF) {
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
const auto OpSize = IROp->Size;
auto Op = IROp->C<IR::IROp_Vector_SToF>();
switch (Op->Header.ElementSize) {
case 4:
scvtf(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
scvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
switch (ElementSize) {
case 2:
scvtf(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
scvtf(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
scvtf(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
break;
}
} else {
switch (ElementSize) {
case 2:
scvtf(Dst.V8H(), Vector.V8H());
break;
case 4:
scvtf(Dst.V4S(), Vector.V4S());
break;
case 8:
scvtf(Dst.V2D(), Vector.V2D());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
break;
}
DEF_OP(Vector_FToZU) {
auto Op = IROp->C<IR::IROp_Vector_FToZU>();
switch (Op->Header.ElementSize) {
case 4:
fcvtzu(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
fcvtzu(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Vector_FToZS) {
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
const auto OpSize = IROp->Size;
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
switch (Op->Header.ElementSize) {
case 4:
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
switch (ElementSize) {
case 2:
fcvtzs(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
fcvtzs(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
fcvtzs(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
break;
}
} else {
switch (ElementSize) {
case 2:
fcvtzs(Dst.V8H(), Vector.V8H());
break;
case 4:
fcvtzs(Dst.V4S(), Vector.V4S());
break;
case 8:
fcvtzs(Dst.V2D(), Vector.V2D());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
break;
}
DEF_OP(Vector_FToU) {
auto Op = IROp->C<IR::IROp_Vector_FToU>();
switch (Op->Header.ElementSize) {
case 4:
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
fcvtzu(GetDst(Node).V4S(), GetDst(Node).V4S());
break;
case 8:
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
fcvtzu(GetDst(Node).V2D(), GetDst(Node).V2D());
break;
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Vector_FToS) {
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
switch (ElementSize) {
case 2:
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
fcvtzs(Dst.Z().VnH(), Mask, Dst.Z().VnH());
break;
case 4:
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
fcvtzs(Dst.Z().VnS(), Mask, Dst.Z().VnS());
break;
case 8:
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
fcvtzs(Dst.Z().VnD(), Mask, Dst.Z().VnD());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
break;
}
} else {
switch (ElementSize) {
case 2:
frinti(Dst.V8H(), Vector.V8H());
fcvtzs(Dst.V8H(), Dst.V8H());
break;
case 4:
frinti(Dst.V4S(), Vector.V4S());
fcvtzs(Dst.V4S(), Dst.V4S());
break;
case 8:
frinti(Dst.V2D(), Vector.V2D());
fcvtzs(Dst.V2D(), Dst.V2D());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
break;
}
auto Op = IROp->C<IR::IROp_Vector_FToS>();
switch (Op->Header.ElementSize) {
case 4:
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
break;
case 8:
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
break;
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Vector_FToF) {
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
const auto OpSize = IROp->Size;
auto Op = IROp->C<IR::IROp_Vector_FToF>();
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
// Curiously, FCVTLT and FCVTNT have no bottom variants,
// and also interesting is that FCVTLT will iterate the
// source vector by accessing each odd element and storing
// them consecutively in the destination.
//
// FCVTNT is somewhat like the opposite. It will read each
// consecutive element, but store each result into every odd
// element in the destination vector.
//
// We need to undo the behavior of FCVTNT with UZP2. In the case
// of FCVTLT, we instead need to set the vector up with ZIP1, so
// that the elements will be processed correctly.
const auto Mask = PRED_TMP_32B.Merging();
switch (Conv) {
case 0x0402: { // Float <- Half
zip1(Dst.Z().VnH(), Vector.Z().VnH(), Vector.Z().VnH());
fcvtlt(Dst.Z().VnS(), Mask, Dst.Z().VnH());
break;
}
case 0x0804: { // Double <- Float
zip1(Dst.Z().VnS(), Vector.Z().VnS(), Vector.Z().VnS());
fcvtlt(Dst.Z().VnD(), Mask, Dst.Z().VnS());
break;
}
case 0x0204: { // Half <- Float
fcvtnt(Dst.Z().VnH(), Mask, Vector.Z().VnS());
uzp2(Dst.Z().VnH(), Dst.Z().VnH(), Dst.Z().VnH());
break;
}
case 0x0408: { // Float <- Double
fcvtnt(Dst.Z().VnS(), Mask, Vector.Z().VnD());
uzp2(Dst.Z().VnS(), Dst.Z().VnS(), Dst.Z().VnS());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
break;
switch (Conv) {
case 0x0804: { // Double <- Float
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2S());
break;
}
} else {
switch (Conv) {
case 0x0402: { // Float <- Half
fcvtl(Dst.V4S(), Vector.V4H());
break;
}
case 0x0804: { // Double <- Float
fcvtl(Dst.V2D(), Vector.V2S());
break;
}
case 0x0204: { // Half <- Float
fcvtn(Dst.V4H(), Vector.V4S());
break;
}
case 0x0408: { // Float <- Double
fcvtn(Dst.V2S(), Vector.V2D());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
break;
}
}
}
DEF_OP(Vector_FToI) {
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (ElementSize) {
case 2:
frintn(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frintn(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frintn(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (ElementSize) {
case 2:
frintm(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frintm(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frintm(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (ElementSize) {
case 2:
frintp(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frintp(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frintp(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
case FEXCore::IR::Round_Towards_Zero.Val:
switch (ElementSize) {
case 2:
frintz(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frintz(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frintz(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
case FEXCore::IR::Round_Host.Val:
switch (ElementSize) {
case 2:
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
}
} else {
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (ElementSize) {
case 2:
frintn(Dst.V8H(), Vector.V8H());
break;
case 4:
frintn(Dst.V4S(), Vector.V4S());
break;
case 8:
frintn(Dst.V2D(), Vector.V2D());
break;
}
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (ElementSize) {
case 2:
frintm(Dst.V8H(), Vector.V8H());
break;
case 4:
frintm(Dst.V4S(), Vector.V4S());
break;
case 8:
frintm(Dst.V2D(), Vector.V2D());
break;
}
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (ElementSize) {
case 2:
frintp(Dst.V8H(), Vector.V8H());
break;
case 4:
frintp(Dst.V4S(), Vector.V4S());
break;
case 8:
frintp(Dst.V2D(), Vector.V2D());
break;
}
break;
case FEXCore::IR::Round_Towards_Zero.Val:
switch (ElementSize) {
case 2:
frintz(Dst.V8H(), Vector.V8H());
break;
case 4:
frintz(Dst.V4S(), Vector.V4S());
break;
case 8:
frintz(Dst.V2D(), Vector.V2D());
break;
}
break;
case FEXCore::IR::Round_Host.Val:
switch (ElementSize) {
case 2:
frinti(Dst.V8H(), Vector.V8H());
break;
case 4:
frinti(Dst.V4S(), Vector.V4S());
break;
case 8:
frinti(Dst.V2D(), Vector.V2D());
break;
}
break;
case 0x0408: { // Float <- Double
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
}
default: LogMan::Msg::A("Unknown Conversion Type : 0%04x", Conv); break;
}
}
@@ -573,13 +203,16 @@ void Arm64JITCore::RegisterConversionHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
REGISTER_OP(VINSGPR, VInsGPR);
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
REGISTER_OP(FLOAT_FROMGPR_U, Float_FromGPR_U);
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
REGISTER_OP(FLOAT_FTOF, Float_FToF);
REGISTER_OP(VECTOR_UTOF, Vector_UToF);
REGISTER_OP(VECTOR_STOF, Vector_SToF);
REGISTER_OP(VECTOR_FTOZU, Vector_FToZU);
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
REGISTER_OP(VECTOR_FTOU, Vector_FToU);
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
#undef REGISTER_OP
}
}
@@ -10,60 +10,61 @@ $end_info$
namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(AESImc) {
auto Op = IROp->C<IR::IROp_VAESImc>();
aesimc(GetDst(Node).V16B(), GetSrc(Op->Vector.ID()).V16B());
aesimc(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
}
DEF_OP(AESEnc) {
auto Op = IROp->C<IR::IROp_VAESEnc>();
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
aese(VTMP1.V16B(), VTMP2.V16B());
aesmc(VTMP1.V16B(), VTMP1.V16B());
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
}
DEF_OP(AESEncLast) {
auto Op = IROp->C<IR::IROp_VAESEncLast>();
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
aese(VTMP1.V16B(), VTMP2.V16B());
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
}
DEF_OP(AESDec) {
auto Op = IROp->C<IR::IROp_VAESDec>();
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
aesd(VTMP1.V16B(), VTMP2.V16B());
aesimc(VTMP1.V16B(), VTMP1.V16B());
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
}
DEF_OP(AESDecLast) {
auto Op = IROp->C<IR::IROp_VAESDecLast>();
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
aesd(VTMP1.V16B(), VTMP2.V16B());
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
}
DEF_OP(AESKeyGenAssist) {
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
aarch64::Literal ConstantLiteral (0x0C030609'0306090CULL, 0x040B0E01'0B0E0104ULL);
aarch64::Label Constant;
aarch64::Label PastConstant;
// Do a "regular" AESE step
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
mov(VTMP1.V16B(), GetSrc(Op->Src.ID()).V16B());
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
aese(VTMP1.V16B(), VTMP2.V16B());
// Do a table shuffle to undo ShiftRows
ldr(VTMP3, &ConstantLiteral);
adr(TMP1.X(), &Constant);
ldr(VTMP3, MemOperand(TMP1.X()));
// Now EOR in the RCON
if (Op->RCON) {
@@ -79,68 +80,24 @@ DEF_OP(AESKeyGenAssist) {
}
b(&PastConstant);
place(&ConstantLiteral);
bind(&Constant);
dc32(0x0B0E0104);
dc32(0x040B0E01);
dc32(0x0306090C);
dc32(0x0C030609);
bind(&PastConstant);
}
DEF_OP(CRC32) {
auto Op = IROp->C<IR::IROp_CRC32>();
switch (Op->SrcSize) {
case 1:
crc32cb(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
break;
case 2:
crc32ch(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
break;
case 4:
crc32cw(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
break;
case 8:
crc32cx(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_64>(Op->Src2.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
}
}
DEF_OP(PCLMUL) {
auto Op = IROp->C<IR::IROp_PCLMUL>();
auto Dst = GetDst(Node).Q();
auto Src1 = GetSrc(Op->Src1.ID()).V2D();
auto Src2 = GetSrc(Op->Src2.ID()).V2D();
switch (Op->Selector) {
case 0b00000000:
pmull(Dst, Src1, Src2);
break;
case 0b00000001:
mov(VTMP1.V1D(), Src1, 1);
pmull(Dst, VTMP1.V2D(), Src2);
break;
case 0b00010000:
mov(VTMP1.V1D(), Src2, 1);
pmull(Dst, VTMP1.V2D(), Src1);
break;
case 0b00010001:
pmull2(Dst, Src1, Src2);
break;
default:
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
break;
}
}
#undef DEF_OP
void Arm64JITCore::RegisterEncryptionHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
REGISTER_OP(VAESIMC, AESImc);
REGISTER_OP(VAESENC, AESEnc);
REGISTER_OP(VAESENCLAST, AESEncLast);
REGISTER_OP(VAESDEC, AESDec);
REGISTER_OP(VAESDECLAST, AESDecLast);
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
REGISTER_OP(CRC32, CRC32);
REGISTER_OP(PCLMUL, PCLMUL);
REGISTER_OP(VAESIMC, AESImc);
REGISTER_OP(VAESENC, AESEnc);
REGISTER_OP(VAESENCLAST, AESEncLast);
REGISTER_OP(VAESDEC, AESDec);
REGISTER_OP(VAESDECLAST, AESDecLast);
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
#undef REGISTER_OP
}
}
@@ -10,10 +10,10 @@ namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(GetHostFlag) {
auto Op = IROp->C<IR::IROp_GetHostFlag>();
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()), Op->Flag, 1);
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), Op->Flag, 1);
}
#undef DEF_OP
File diff suppressed because it is too large. Load diff
Loaded 100 of 994 files, more files were not shown because too many files have changed in this diff. Show more