Compare commits

..
1 Commits
Author SHA1 Message Date
Ryan Houdek 33fe6813fc Docs: Update for release FEX-2104 2021-04-02 11:35:29 -07:00
1296 changed files with 40676 additions and 162411 deletions

No files matched your search

@@ -1,45 +0,0 @@
---
name: Potential Game Bug
about: A bug in FEX-Emu that causes a problem in a game
title: "[Game]: [Short Problem Description]"
labels: Game related
assignees: ''
---
**What Game**
The game name.
A link to the storefront where to get the game. GOG, Steam, Itch.io, etc
**Describe the bug**
A clear and concise description of what the bug is.
**To Reproduce**
Steps to reproduce the behavior:
1. Go to '...'
2. Click on '....'
3. Scroll down to '....'
4. See error
**Expected behavior**
A clear and concise description of what you expected to happen.
**Screenshots and Video**
If applicable, add screenshots and video to help explain your problem.
**System information:**
- OS: [eg: Ubuntu 21.10]
- CPU/SoC: [eg: Snapdragon 888, Intel Core i8-12900k]
- Video driver version: [eg: OpenGL ES 3.2 Mesa 22.0.0-devel (git-9ff086052a)]
- RootFS used: [eg: Ubuntu 21.10 Official Rootfs]
- FEX version: (FEXGetConfig --version) [eg: FEX-2112-155-gc691d709]
- Thunks Enabled: [Yes/No]
**Additional context**
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
- Is this a Vulkan game: [Yes/No/Unknown]
- If Yes, What is your Vulkan driver:
Add any other context about the problem here.
+4 -109
View File
@@ -13,14 +13,13 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
jobs:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2], [self-hosted, ARMv8.4]]
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2]]
fail-fast: false
steps:
@@ -29,24 +28,9 @@ jobs:
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
run: git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
@@ -64,7 +48,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -132,18 +116,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
- name: gcc target tests 32
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
- name: GCC32 Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
- name: Struct verifier tests
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -155,84 +127,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
- name: APITest tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target api_tests
- name: APITest Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
- name: ARMEmitter tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target emitter_tests
- name: ARMEmitter Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ARMEmitterTests.log || true
- name: FEXLinuxTests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
- name: FEXLinuxTests Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
- name: Thunkgen tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target thunkgen_tests
- name: Thunkgen Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
- name: Install
if: matrix.arch[1] == 'x64'
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: Test GL No-Thunks
if: matrix.arch[1] == 'x64'
working-directory: ${{runner.workspace}}/build
shell: bash
env:
DISPLAY: ":0"
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_nothunks
- name: No thunks Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_NoThunkResults.log || true
- name: Test GL Thunks
if: matrix.arch[1] == 'x64'
working-directory: ${{runner.workspace}}/build
shell: bash
env:
DISPLAY: ":0"
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_thunks
- name: Thunks Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkResults.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
@@ -252,3 +146,4 @@ jobs:
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
-118
View File
@@ -1,118 +0,0 @@
name: Vixl Simulator run
on:
push:
branches:
- main
pull_request:
branches:
- main
env:
# Customize the CMake build type here (Release, Debug, RelWithDebInfo, etc.)
BUILD_TYPE: Release
CC: clang
CXX: clang++
jobs:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
# Only the x86-64 runner is fast enough to run this
arch: [[self-hosted, x64], [self-hosted, ARMv8.4]]
fail-fast: false
steps:
- uses: actions/checkout@v2
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: 'actions/upload-artifact@v2'
with:
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
-1
View File
@@ -10,4 +10,3 @@ out/
.vscode/
.vs/
*.pyc
.cache
+1 -24
View File
@@ -1,7 +1,7 @@
[submodule "External/vixl"]
shallow = true
path = External/vixl
url = https://github.com/FEX-Emu/vixl.git
url = https://github.com/Sonicadvance1/vixl.git
[submodule "External/cpp-optparse"]
path = External/cpp-optparse
url = https://github.com/Sonicadvance1/cpp-optparse
@@ -30,26 +30,3 @@
shallow = true
path = External/fex-gcc-target-tests-bins
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
[submodule "External/jemalloc"]
path = External/jemalloc
url = https://github.com/FEX-Emu/jemalloc.git
[submodule "External/fmt"]
path = External/fmt
url = https://github.com/fmtlib/fmt.git
[submodule "External/drm-headers"]
path = External/drm-headers
url = https://github.com/FEX-Emu/drm-headers.git
[submodule "External/xxhash"]
path = External/xxhash
url = https://github.com/FEX-Emu/xxHash.git
[submodule "External/Catch2"]
path = External/Catch2
url = https://github.com/catchorg/Catch2.git
[submodule "External/robin-map"]
shallow = true
path = External/robin-map
url = https://github.com/Tessil/robin-map.git
[submodule "External/Vulkan-Headers"]
shallow = true
path = External/Vulkan-Headers
url = https://github.com/KhronosGroup/Vulkan-Headers.git
-5
View File
@@ -1,5 +0,0 @@
{
"ThunksDB": {
"GL": 1
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"ThunksDB": {
"Vulkan": 1
}
}
-21
View File
@@ -1,21 +0,0 @@
if(NOT EXISTS "@CMAKE_BINARY_DIR@/install_manifest.txt")
message(FATAL_ERROR "Cannot find install manifest: @CMAKE_BINARY_DIR@/install_manifest.txt")
endif()
file(READ "@CMAKE_BINARY_DIR@/install_manifest.txt" files)
string(REGEX REPLACE "\n" ";" files "${files}")
foreach(file ${files})
message(STATUS "Uninstalling $ENV{DESTDIR}${file}")
if(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
exec_program(
"@CMAKE_COMMAND@" ARGS "-E remove \"$ENV{DESTDIR}${file}\""
OUTPUT_VARIABLE rm_out
RETURN_VALUE rm_retval
)
if(NOT "${rm_retval}" STREQUAL 0)
message(FATAL_ERROR "Problem when removing $ENV{DESTDIR}${file}")
endif()
else(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
message(STATUS "File $ENV{DESTDIR}${file} does not exist.")
endif()
endforeach()
+73 -347
View File
@@ -1,70 +1,24 @@
cmake_minimum_required(VERSION 3.14)
project(FEX)
INCLUDE (CheckIncludeFiles)
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
option(BUILD_THUNKS "Build thunks" FALSE)
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
option(ENABLE_IWYU "Enables include what you use program" FALSE)
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
option(ENABLE_LLD "Enable linking with lld" FALSE)
option(ENABLE_MOLD "Enable linking with mold" FALSE)
option(ENABLE_LLD "Enable linking with LLD" FALSE)
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
option(ENABLE_WERROR "Enables -Werror" FALSE)
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
if (ENABLE_FEXCORE_PROFILER)
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
else()
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
endif()
endif()
# uninstall target
if(NOT TARGET uninstall)
configure_file(
"${CMAKE_CURRENT_SOURCE_DIR}/CMakeFiles/cmake_uninstall.cmake.in"
"${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake"
IMMEDIATE @ONLY)
add_custom_target(uninstall
COMMAND ${CMAKE_COMMAND} -P ${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake)
endif()
# These options are meant for package management
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
set (OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version in the format of <MMYY>{.<REV>}")
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
if (CMAKE_BUILD_TYPE MATCHES "DEBUG")
set(ENABLE_ASSERTIONS TRUE)
@@ -75,17 +29,6 @@ if (ENABLE_ASSERTIONS)
add_definitions(-DASSERTIONS_ENABLED=1)
endif()
if (ENABLE_GDB_SYMBOLS)
message(STATUS "GDBSymbols support enabled")
add_definitions(-DGDB_SYMBOLS_ENABLED=1)
endif()
if (ENABLE_INTERPRETER)
message(STATUS "Interpreter enabled")
add_definitions(-DINTERPRETER_ENABLED=1)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
@@ -101,6 +44,38 @@ else()
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION FALSE)
endif()
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
message(STATUS "CCache enabled")
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
endif()
if (ENABLE_XRAY)
add_compile_options(-fxray-instrument)
link_libraries(-fxray-instrument)
endif()
if (ENABLE_LLD)
link_libraries(-fuse-ld=lld)
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address)
link_libraries(-fno-omit-frame-pointer -fsanitize=address)
endif()
if (ENABLE_TSAN)
add_compile_options(-fno-omit-frame-pointer -fsanitize=thread)
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
endif()
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
@@ -115,94 +90,11 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
set(_M_ARM_64 1)
add_definitions(-D_M_ARM_64=1)
endif()
if (ENABLE_CCACHE)
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
message(STATUS "CCache enabled")
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
endif()
endif()
if (ENABLE_XRAY)
add_compile_options(-fxray-instrument)
link_libraries(-fxray-instrument)
endif()
if (ENABLE_COMPILE_TIME_TRACE)
add_compile_options(-ftime-trace)
link_libraries(-ftime-trace)
endif()
set (PTHREAD_LIB pthread)
if (ENABLE_LLD AND ENABLE_MOLD)
message (FATAL_ERROR "Cannot enable both lld and mold")
elseif (ENABLE_LLD)
set (LD_OVERRIDE "-fuse-ld=lld")
add_link_options(${LD_OVERRIDE})
elseif (ENABLE_MOLD)
add_link_options("-fuse-ld=mold")
endif()
if (ENABLE_LIBCXX)
message(WARNING "This is an unsupported configuration and should only be used for testing")
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -stdlib=libc++")
set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -stdlib=libc++ -lc++abi")
endif()
if (NOT ENABLE_OFFLINE_TELEMETRY)
# Disable FEX offline telemetry entirely if asked
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
endif()
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
add_definitions(-DTERMUX_BUILD=1)
set(TERMUX_BUILD 1)
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
set(ENABLE_JEMALLOC FALSE)
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
link_libraries(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
endif()
if (ENABLE_TSAN)
add_compile_options(-fno-omit-frame-pointer -fsanitize=thread)
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
endif()
if (ENABLE_JEMALLOC)
add_definitions(-DENABLE_JEMALLOC=1)
add_subdirectory(External/jemalloc/)
include_directories(External/jemalloc/pregen/include/)
else()
message (STATUS
" jemalloc disabled!\n"
" This is not a recommended configuration!\n"
" This will very explicitly break 32-bit application execution!\n"
" Use at your own risk!")
endif()
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
include_directories(External/robin-map/include/)
if (BUILD_TESTS)
# Enable vixl disassembler if tests are enabled.
set(COMPILE_VIXL_DISASSEMBLER TRUE)
endif()
add_subdirectory(External/vixl/)
include_directories(External/vixl/src/)
@@ -211,30 +103,14 @@ if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
add_subdirectory(External/xxhash/)
include_directories(External/xxhash/)
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
if (BUILD_TESTS)
option(CATCH_BUILD_STATIC_LIBRARY "" ON)
set(CATCH_BUILD_STATIC_LIBRARY ON)
add_subdirectory(External/Catch2/)
# Pull in catch_discover_tests definition
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
include(Catch)
endif()
add_subdirectory(External/cpp-optparse/)
include_directories(External/cpp-optparse/)
add_subdirectory(External/fmt/)
add_subdirectory(External/imgui/)
include_directories(External/imgui/)
@@ -268,6 +144,11 @@ if(ENUM_ENUM_WARNING)
add_compile_options(-Wno-deprecated-enum-enum-conversion)
endif()
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
if(COMPILER_SUPPORTS_MARCH_NATIVE)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -march=native")
endif()
if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
add_compile_options(-Werror)
if (NOT ENABLE_STRICT_WERROR)
@@ -276,51 +157,19 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
endif()
endif()
if (NOT TUNE_ARCH STREQUAL "generic")
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
if(COMPILER_SUPPORTS_ARCH_TYPE)
add_compile_options("-march=${TUNE_ARCH}")
else()
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
endif()
endif()
if(_M_ARM_64)
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
# Manually detect newer CPU revisions until clang and llvm fixes their bug
# This script will either provide a supported CPU or 'native'
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo"
OUTPUT_VARIABLE AARCH64_CPU)
if (TUNE_CPU STREQUAL "native")
if(_M_ARM_64)
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
add_compile_options("-mcpu=native")
endif()
else()
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
# Manually detect newer CPU revisions until clang and llvm fixes their bug
# This script will either provide a supported CPU or 'native'
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
OUTPUT_VARIABLE AARCH64_CPU)
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
add_compile_options("-mcpu=${AARCH64_CPU}")
endif()
endif()
else()
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
if(COMPILER_SUPPORTS_MARCH_NATIVE)
add_compile_options("-march=native")
endif()
endif()
else()
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
add_compile_options("-mcpu=${TUNE_CPU}")
else()
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=${AARCH64_CPU}")
endif()
endif()
@@ -387,181 +236,58 @@ add_compile_options(-Wall)
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/include/Config.h.in
${CMAKE_BINARY_DIR}/generated/ConfigDefines.h)
${CMAKE_BINARY_DIR}/generated/Config.h)
if (BUILD_TESTS)
include(CTest)
enable_testing()
message(STATUS "Unit tests are enabled")
endif()
add_subdirectory(FEXHeaderUtils/)
add_subdirectory(External/FEXCore)
# Binfmt_misc files must be installed prior to Source/ installs
add_subdirectory(Data/binfmts/)
add_subdirectory(Source/)
add_subdirectory(Data/AppConfig/)
# Install the ThunksDB file
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/)
endforeach()
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
if (BUILD_THUNKS)
set (FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
add_subdirectory(ThunkLibs/Generator)
# Thunk targets for both host libraries and IDE integration
add_subdirectory(ThunkLibs/HostLibs)
# Thunk targets for IDE integration of guest code, only
add_subdirectory(ThunkLibs/GuestLibs)
# Thunk targets for guest libraries
include(ExternalProject)
ExternalProject_Add(host-libs
PREFIX host-libs
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/HostLibs"
BINARY_DIR "Host"
CMAKE_ARGS "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
)
install(
CODE "MESSAGE(\"-- Installing: host-libs\")"
CODE "
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkHostsInstall
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Host
)"
DEPENDS host-libs
)
ExternalProject_Add(guest-libs
PREFIX guest-libs
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
BINARY_DIR "Guest"
CMAKE_ARGS
"-DBITNESS=64"
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_64_TOOLCHAIN_FILE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
CMAKE_ARGS "-DX86_C_COMPILER:STRING=${X86_C_COMPILER}" "-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}" "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
DEPENDS thunkgen
)
ExternalProject_Add(guest-libs-32
PREFIX guest-libs-32
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
BINARY_DIR "Guest_32"
CMAKE_ARGS
"-DBITNESS=32"
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_32_TOOLCHAIN_FILE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
DEPENDS thunkgen
)
install(
CODE "MESSAGE(\"-- Installing: guest-libs\")"
CODE "
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkGuestsInstall
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)"
DEPENDS guest-libs
)
install(
CODE "MESSAGE(\"-- Installing: guest-libs-32\")"
CODE "
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)"
DEPENDS guest-libs-32
)
add_custom_target(uninstall_guest-libs
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)
add_custom_target(uninstall_guest-libs-32
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)
add_dependencies(uninstall uninstall_guest-libs)
add_dependencies(uninstall uninstall_guest-libs-32)
endif()
set(FEX_VERSION_MAJOR "0")
set(FEX_VERSION_MINOR "0")
set(FEX_VERSION_PATCH "0")
if (OVERRIDE_VERSION STREQUAL "detect")
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
RESULT_VARIABLE GIT_ERROR
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
if (NOT ${GIT_ERROR} EQUAL 0)
# Likely built in a way that doesn't have tags
# Setup a version tag that is unknown
set(GIT_DESCRIBE_STRING "FEX-0000")
endif()
endif()
else()
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
endif()
# Parse the version here
# Change something like `FEX-2106.1-76-<hash>` in to a list
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
# Extract the `2106.1` element
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
# Change `2106.1` in to a list
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
# Calculate list size
list(LENGTH DESCRIBE_LIST LIST_SIZE)
# Pull out the major version
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
# Minor version only exists if there is a .1 at the end
# eg: 2106 versus 2106.1
if (LIST_SIZE GREATER 1)
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
endif()
# Package creation
set (CPACK_GENERATOR "DEB")
set (CPACK_PACKAGE_NAME fex-emu)
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/CPack/Description.txt")
# Debian defines
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
"${CMAKE_CURRENT_SOURCE_DIR}/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/CPack/triggers")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
# binfmt_misc conflicts with qemu-user-static
# We also only install binfmt_misc on aarch64 hosts
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
endif()
include (CPack)
+1 -1
View File
@@ -55,7 +55,7 @@ further defined and clarified by project maintainers.
## Enforcement
Instances of abusive, harassing, or otherwise unacceptable behavior may be
reported by contacting the project team at team@fex-emu.com. All
reported by contacting the project team at team@fex-emu.org. All
complaints will be reviewed and investigated and will result in a response that
is deemed necessary and appropriate to the circumstances. The project team is
obligated to maintain confidentiality with regard to the reporter of an incident.
-3
View File
@@ -1,3 +0,0 @@
x86 and x86-64 Linux emulator
FEX is very much work in progress, so expect things to change.
-18
View File
@@ -1,18 +0,0 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Setup binfmt_misc
update-binfmts --import FEX-x86
update-binfmts --import FEX-x86_64
}
# Install FEXInterpreter hardlink
# Needs to be done before setting up binfmt_misc
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
-17
View File
@@ -1,17 +0,0 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Uninstall
update-binfmts --unimport FEX-x86
update-binfmts --unimport FEX-x86_64
}
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
# Remove FEXInterpreter hardlink
unlink /usr/bin/FEXInterpreter
-1
View File
@@ -1 +0,0 @@
activate-noawait ldconfig
+3 -3
View File
@@ -11,15 +11,15 @@ endforeach()
# First generate then install it
foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
# Get the filename only component
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WLE)
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WE)
# Configure it
configure_file(
${GEN_CONFIG_SRC}
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json)
# Then install the configured json
install(
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"x86dec_SynchronizeRIPOnAllBlocks": "1"
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"x86dec_SynchronizeRIPOnAllBlocks": "1"
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"AdditionalArguments": "--no-sandbox"
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"x86dec_SynchronizeRIPOnAllBlocks": "1"
}
}
-5
View File
@@ -1,5 +0,0 @@
{
"Config": {
"x86dec_SynchronizeRIPOnAllBlocks": "1"
}
}
-178
View File
@@ -1,178 +0,0 @@
{
"DB": {
"GL": {
"Library" : "libGL-guest.so",
"Depends": [
"X11"
],
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.2.0",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.7.0"
]
},
"GLESv2": {
"Library": "libGLESv2-guest.so",
"Depends": [
"X11"
],
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2.0.0"
]
},
"X11": {
"Library": "libX11-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6.4.0"
]
},
"Vulkan": {
"Library": "libvulkan-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so.1",
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
],
"Comment": [
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
]
},
"xcb": {
"Library": "libxcb-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1.1.0"
]
},
"xcb-dri2": {
"Library": "libxcb-dri2-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0.0.0"
]
},
"xcb-dri3": {
"Library": "libxcb-dri3-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0.0.0"
]
},
"xcb-xfixes": {
"Library": "libxcb-xfixes-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0.0.0"
]
},
"xcb-shm": {
"Library": "libxcb-shm-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0.0.0"
]
},
"xcb-sync": {
"Library": "libxcb-sync-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1.0.0"
]
},
"xcb-randr": {
"Library": "libxcb-randr-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0.1.0"
]
},
"xcb-present": {
"Library": "libxcb-present-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0.0.0"
]
},
"xcb-glx": {
"Library": "libxcb-glx-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0.0.0"
]
},
"xshmfence": {
"Library": "libxshmfence-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1.0.0"
]
},
"drm": {
"Library": "libdrm-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2.4.0"
]
},
"asound": {
"Library": "libasound-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2.0.0"
]
},
"Xrender": {
"Library": "libXrender-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1.3.0"
]
},
"Xext": {
"Library": "libXext-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6.4.0"
]
},
"Xfixes": {
"Library": "libXfixes-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3.1.0"
]
},
"OpenCL": {
"Library" : "libOpenCL-guest.so",
"Overlay": [
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1",
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1.0.0"
]
},
"":{}
}
}
-17
View File
@@ -1,17 +0,0 @@
function(GenBinFmt Name)
# Get the filename only component
get_filename_component(FMT_NAME ${Name} NAME_WE)
# Configure it
configure_file(
${Name}
${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
# Then install the configured binfmt
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
endfunction()
GenBinFmt(FEX-x86.in)
GenBinFmt(FEX-x86_64.in)
-8
View File
@@ -1,8 +0,0 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
credentials yes
fix_binary yes
preserve yes
-8
View File
@@ -1,8 +0,0 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
credentials yes
fix_binary yes
preserve yes
+2 -2
View File
@@ -3,7 +3,7 @@ FROM ubuntu:20.04 as builder
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
clang-10 llvm-10 nasm ninja-build \
clang-10 llvm-10 nasm ninja-build libnuma-dev \
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
python3 linux-headers-generic
@@ -23,7 +23,7 @@ FROM ubuntu:20.04
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y \
libcap-dev libglfw3-dev libepoxy-dev
libnuma-dev libcap-dev libglfw3-dev libepoxy-dev
COPY --from=builder /opt/FEX/build/Bin/* /usr/bin/
-1
Submodule External/Catch2 deleted from c4e3767e26.
+23 -37
View File
@@ -9,20 +9,14 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
set(_M_ARM_64 1)
endif()
if (ENABLE_VIXL_SIMULATOR)
# If the vixl simulator is enabled then we are using the ARM64 JIT
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" FALSE)
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" TRUE)
else()
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" ${_M_X86_64})
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" ${_M_ARM_64})
endif()
set(ENABLE_JIT_X86_64 ${_M_X86_64} CACHE BOOL "Enable the x86_64 JIT")
set(ENABLE_JIT_ARM64 ${_M_ARM_64} CACHE BOOL "Enable the ARM64 JIT")
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
option(ENABLE_JITSYMBOLS "Enable visibility of JITSymbols in profiling tools" FALSE)
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
@@ -34,6 +28,7 @@ set(CMAKE_INCLUDE_CURRENT_DIR ON)
include(CheckCXXCompilerFlag)
include(CheckIncludeFileCXX)
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
# Useful to have for freestanding libFEXCore
add_subdirectory(External/vixl/)
@@ -43,32 +38,27 @@ endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
# Find our git hash
find_package(Git)
set(GIT_SHORT_HASH "Unknown")
set(GIT_DESCRIBE_STRING "FEX-Unknown")
if (OVERRIDE_VERSION STREQUAL "detect")
# Find our git hash
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} rev-parse --short=7 HEAD
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_SHORT_HASH
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
endif()
else()
set(GIT_SHORT_HASH "${OVERRIDE_VERSION}")
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} rev-parse --short HEAD
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_SHORT_HASH
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
endif()
configure_file(
@@ -82,7 +72,3 @@ add_subdirectory(Source/)
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
DESTINATION include
COMPONENT Development)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
+12 -2
View File
@@ -18,8 +18,14 @@ This project aims to provide a fast and functional x86-64 emulation library that
* Portable library implementation in order to support easy integration in to applications
### Target Host Architecture
The target host architecture for this library is AArch64. Specifically the ARMv8.1 version or newer.
The CPU IR is designed with AArch64 in mind but should allow for other architectures as well.
x86-64 host support is available for ease of development, but is not a priority.
The CPU IR is designed with AArch64 in mind but there is a desire to run the recompiled code on other architectures as well.
Multiple architecture support is desired for easier bringup and debugging, performance isn't as much of a priority there (ex. x86-64(guest) translated to x86-64(host))
### Not currently goals but will be in the future
* 32bit x86 support
* This will be a desire in the future, but to lower the amount of work required, decided to push this off for now.
* Integration in to WINE
* Later generation of x86-64 instruction sets
* Including AVX, F16C, XOP, FMA, AVX2, etc
### Not desired
* Kernel space emulation
* CPL0-2 emulation
@@ -27,3 +33,7 @@ x86-64 host support is available for ease of development, but is not a priority.
* IRQs
* SVM
* "Cycle Accurate" emulation
### Dependencies
* clang-tidy if you want to ensure the code stays tidy
* cmake
* A C++17 compliant compiler (There are assumptions made about using Clang and LTO)
+5 -65
View File
@@ -98,19 +98,14 @@ def print_man_option(short, long, desc, default):
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
def print_man_env_option(name, desc, default, no_json_key):
output_man.write("\\fBFEX_{0}\\fR\n".format(name.upper()))
def print_man_env_option(name, desc, default):
output_man.write("\\fBFEX_{0}\\fR\n".format(name))
# Print description
for line in desc:
output_man.write(".Pp\n")
output_man.write("{0}\n".format(line))
if (not no_json_key):
output_man.write(".Pp\n")
output_man.write("\\fBJSON key:\\fR '{0}'\n".format(name))
output_man.write(".Pp\n\n")
output_man.write(".Pp\n")
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
@@ -159,48 +154,12 @@ def print_man_environment(options):
# Wrap the string argument in quotes
default = "'" + default + "'"
print_man_env_option(
op_key,
op_key.upper(),
op_vals["Desc"],
default,
False
default
)
print_man_environment_tail()
output_man.write(".El\n")
def print_man_environment_tail():
# Additional environment variables that live outside of the normal loop
print_man_env_option(
"FEX_APP_CONFIG_LOCATION",
[
"Allows the user to override where FEX looks for configuration files",
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
"This will override the full path",
],
"''", True)
print_man_env_option(
"FEX_APP_CONFIG",
[
"Allows the user to override where FEX looks for only the application config file",
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/Config.json",
"This will override this file location",
"One must be careful with this option as it will override any applications that load with execve as well"
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
],
"''", True)
print_man_env_option(
"FEX_APP_DATA_LOCATION",
[
"Allows the user to override where FEX looks for data files",
"By default FEX will look in {$HOME, $XDG_DATA_HOME}/.fex-emu/",
"This will override the full path",
"This is the folder where FEX stores generated files like IR cache"
],
"''", True)
def print_man_header():
header ='''.Dd {0}
.Dt FEX
@@ -374,7 +333,7 @@ def print_parse_argloader_options(options):
conversion_func = "std::to_string"
if ("ArgumentHandler" in op_vals):
NeedsString = True
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
conversion_func = "FEX::Handler::{0}".format(op_vals["ArgumentHandler"])
if (value_type == "str"):
NeedsString = True
conversion_func = ""
@@ -396,21 +355,6 @@ def print_parse_argloader_options(options):
output_argloader.write("#endif\n")
def print_parse_envloader_options(options):
output_argloader.write("#ifdef ENVLOADER\n")
output_argloader.write("#undef ENVLOADER\n")
output_argloader.write("if (false) {}\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
if ("ArgumentHandler" in op_vals):
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
output_argloader.write("Value = {0}(Value);\n".format(conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def check_for_duplicate_options(options):
short_map = []
long_map = []
@@ -485,8 +429,4 @@ output_man.close()
output_argloader = open(output_argumentloader_filename, "w")
print_argloader_options(options);
print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
output_argloader.close()
+43 -6
View File
@@ -7,12 +7,17 @@ OpClasses = collections.OrderedDict()
def get_ir_classes(ops, defines):
global OpClasses
for op_class, opslist in ops.items():
if not (op_class in OpClasses):
OpClasses[op_class] = []
for op_key, op_vals in ops.items():
if not ("Last" in op_vals):
OpClass = "#Unknown"
for op, op_val in opslist.items():
OpClasses[op_class].append([op, op_val])
if ("OpClass" in op_vals):
OpClass = op_vals["OpClass"]
if not (OpClass in OpClasses):
OpClasses[OpClass] = []
OpClasses[OpClass].append([op_key, op_vals])
# Sort the dictionary after we are done parsing it
OpClasses = collections.OrderedDict(sorted(OpClasses.items()))
@@ -33,9 +38,41 @@ def print_ir_ops():
op_key = op[0]
op_vals = op[1]
output_file.write("## %s\n" % (op_key))
HasDest = ("HasDest" in op_vals and op_vals["HasDest"] == True)
HasSSAArgs = ("SSAArgs" in op_vals and len(op_vals["SSAArgs"]) > 0)
HasSSAArgNames = "SSANames" in op_vals
HasArgs = "Args" in op_vals
SSAArgsCount = 0
ArgCount = 0
if (HasSSAArgs):
SSAArgsCount = int(op_vals["SSAArgs"])
if (HasArgs):
ArgCount = len(op_vals["Args"])
TotalArgsCount = SSAArgsCount + (ArgCount / 2)
output_file.write(">")
output_file.write(op_key)
if (HasDest):
output_file.write("%dest = ")
output_file.write("%s " % op_key)
ArgComma = (", ", "")
if (HasSSAArgs):
for i in range(0, SSAArgsCount):
FinalArg = (i + 1) == TotalArgsCount
if (HasSSAArgNames):
output_file.write("%%%s%s" % (op_vals["SSANames"][i], ArgComma[FinalArg]))
else:
output_file.write("%%ssa%d%s" % (i, ArgComma[FinalArg]))
if (HasArgs):
Args = op_vals["Args"]
for i in range(0, ArgCount, 2):
FinalArg = ((i / 2) + SSAArgsCount + 1) == TotalArgsCount
data_type = Args[i]
data_name = Args[i + 1]
output_file.write("\<%s %s\>%s" % (data_type, data_name, ArgComma[FinalArg]))
output_file.write("\n\n")
Vendored Executable → Regular
+402 -526
View File
File diff suppressed because it is too large. Load diff
+38 -120
View File
@@ -1,15 +1,9 @@
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
set (FEXCORE_BASE_SRCS
Common/Paths.cpp
Interface/Config/Config.cpp
Utils/FileLoading.cpp
Utils/ForcedAssert.cpp
Utils/LogManager.cpp
)
set (SRCS
Common/Paths.cpp
Common/JitSymbols.cpp
Common/NetStream.cpp
Common/SoftFloat-3e/extF80_add.c
Common/SoftFloat-3e/extF80_div.c
Common/SoftFloat-3e/extF80_sub.c
@@ -77,25 +71,17 @@ set (SRCS
Common/SoftFloat-3e/f32_to_extF80.c
Common/SoftFloat-3e/s_normSubnormalF32Sig.c
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
Interface/Config/Config.cpp
Interface/Context/Context.cpp
Interface/Core/LookupCache.cpp
Interface/Core/BlockSamplingData.cpp
Interface/Core/CompileService.cpp
Interface/Core/Core.cpp
Interface/Core/CPUBackend.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/GdbServer.cpp
Interface/Core/HostFeatures.cpp
Interface/Core/ObjectCache/JobHandling.cpp
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
Interface/Core/ObjectCache/ObjectCacheService.cpp
Interface/Core/OpcodeDispatcher/Crypto.cpp
Interface/Core/OpcodeDispatcher/Flags.cpp
Interface/Core/OpcodeDispatcher/Vector.cpp
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher/X87F64.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/SignalDelegator.cpp
Interface/Core/X86Tables.cpp
Interface/Core/X86DebugInfo.cpp
Interface/Core/X86HelperGen.cpp
@@ -104,7 +90,8 @@ set (SRCS
Interface/Core/Dispatcher/Dispatcher.cpp
Interface/Core/Dispatcher/X86Dispatcher.cpp
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
Interface/Core/Interpreter/InterpreterFallbacks.cpp
Interface/Core/Interpreter/InterpreterCore.cpp
Interface/Core/Interpreter/InterpreterOps.cpp
Interface/Core/X86Tables/BaseTables.cpp
Interface/Core/X86Tables/DDDTables.cpp
Interface/Core/X86Tables/EVEXTables.cpp
@@ -118,8 +105,6 @@ set (SRCS
Interface/Core/X86Tables/X87Tables.cpp
Interface/Core/X86Tables/XOPTables.cpp
Interface/HLE/Thunks/Thunks.cpp
Interface/GDBJIT/GDBJIT.cpp
Interface/IR/AOTIR.cpp
Interface/IR/IRDumper.cpp
Interface/IR/IRParser.cpp
Interface/IR/IREmitter.cpp
@@ -129,45 +114,25 @@ set (SRCS
Interface/IR/Passes/DeadContextStoreElimination.cpp
Interface/IR/Passes/IRCompaction.cpp
Interface/IR/Passes/IRValidation.cpp
Interface/IR/Passes/RAValidation.cpp
Interface/IR/Passes/LongDivideRemovalPass.cpp
Interface/IR/Passes/ValueDominanceValidation.cpp
Interface/IR/Passes/PhiValidation.cpp
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
Interface/IR/Passes/DeadStoreElimination.cpp
Interface/IR/Passes/StaticRegisterAllocationPass.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/SyscallOptimization.cpp
Utils/Allocator.cpp
Utils/Allocator/64BitAllocator.cpp
Utils/NetStream.cpp
Utils/Telemetry.cpp
Utils/ELFLoader.cpp
Utils/ELFSymbolDatabase.cpp
Utils/LogManager.cpp
Utils/Threads.cpp
Utils/Profiler.cpp
)
if (ENABLE_INTERPRETER)
list(APPEND SRCS
Interface/Core/Interpreter/InterpreterCore.cpp
Interface/Core/Interpreter/InterpreterOps.cpp
Interface/Core/Interpreter/ALUOps.cpp
Interface/Core/Interpreter/AtomicOps.cpp
Interface/Core/Interpreter/BranchOps.cpp
Interface/Core/Interpreter/ConversionOps.cpp
Interface/Core/Interpreter/EncryptionOps.cpp
Interface/Core/Interpreter/F80Ops.cpp
Interface/Core/Interpreter/FlagOps.cpp
Interface/Core/Interpreter/MemoryOps.cpp
Interface/Core/Interpreter/MiscOps.cpp
Interface/Core/Interpreter/MoveOps.cpp
Interface/Core/Interpreter/VectorOps.cpp)
endif()
if(_M_ARM_64)
list(APPEND SRCS
Interface/Core/ArchHelpers/Arm64.cpp)
endif()
set(DEFINES -DTHREAD_LOCAL=_Thread_local)
set(DEFINES )
if (_M_X86_64)
list(APPEND DEFINES -D_M_X86_64=1)
@@ -177,15 +142,6 @@ if (_M_ARM_64)
list(APPEND DEFINES -D_M_ARM_64=1)
endif()
if (ENABLE_VIXL_SIMULATOR)
# We can run the simulator on both x86-64 or AArch64 hosts
list(APPEND DEFINES -DVIXL_SIMULATOR=1 -DVIXL_INCLUDE_SIMULATOR_AARCH64=1)
endif()
if (ENABLE_VIXL_DISASSEMBLER)
list(APPEND DEFINES -DVIXL_DISASSEMBLER=1)
endif()
if (ENABLE_JIT_X86_64)
list(APPEND SRCS
Interface/Core/JIT/x86_64/JIT.cpp
@@ -198,9 +154,7 @@ if (ENABLE_JIT_X86_64)
Interface/Core/JIT/x86_64/MemoryOps.cpp
Interface/Core/JIT/x86_64/MiscOps.cpp
Interface/Core/JIT/x86_64/MoveOps.cpp
Interface/Core/JIT/x86_64/VectorOps.cpp
Interface/Core/JIT/x86_64/x64Relocations.cpp
)
Interface/Core/JIT/x86_64/VectorOps.cpp)
list(APPEND DEFINES -DJIT_X86_64)
endif()
@@ -217,31 +171,25 @@ if (ENABLE_JIT_ARM64)
Interface/Core/JIT/Arm64/MemoryOps.cpp
Interface/Core/JIT/Arm64/MiscOps.cpp
Interface/Core/JIT/Arm64/MoveOps.cpp
Interface/Core/JIT/Arm64/VectorOps.cpp
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
)
Interface/Core/JIT/Arm64/VectorOps.cpp)
endif()
set (LIBS fmt::fmt vixl dl xxhash tiny-json FEXHeaderUtils)
if (ENABLE_JEMALLOC)
list (APPEND LIBS FEX_jemalloc)
if (ENABLE_JITSYMBOLS)
list(APPEND DEFINES -DENABLE_JITSYMBOLS=1)
endif()
# Generate config
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
${CMAKE_BINARY_DIR}/generated/Config/Config.json)
# Generate IR include file
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
add_custom_target(CREATE_IR_FOLDER ALL
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_IR_FOLDER}")
add_custom_command(
OUTPUT "${OUTPUT_NAME}"
DEPENDS "${INPUT_NAME}"
DEPENDS CREATE_IR_FOLDER
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
)
@@ -255,6 +203,7 @@ set(OUTPUT_IR_DOC "${CMAKE_BINARY_DIR}/IR.md")
add_custom_command(
OUTPUT "${OUTPUT_IR_DOC}"
DEPENDS "${INPUT_NAME}"
DEPENDS CREATE_IR_FOLDER
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py" "${INPUT_NAME}" "${OUTPUT_IR_DOC}"
)
@@ -271,28 +220,23 @@ add_custom_target(IR_INC
set(OUTPUT_CONFIG_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/Config")
set(OUTPUT_CONFIG_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigValues.inl")
set(OUTPUT_CONFIG_OPTION_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigOptions.inl")
set(INPUT_CONFIG_NAME "${CMAKE_BINARY_DIR}/generated/Config/Config.json")
set(INPUT_CONFIG_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json")
set(OUTPUT_MAN_NAME "${CMAKE_BINARY_DIR}/generated/FEX.1")
set(OUTPUT_MAN_NAME_COMPRESS "${CMAKE_BINARY_DIR}/generated/FEX.1.gz")
file(MAKE_DIRECTORY "${OUTPUT_CONFIG_FOLDER}")
add_custom_target(CREATE_CONFIG_FOLDER ALL
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_CONFIG_FOLDER}")
add_custom_command(
OUTPUT "${OUTPUT_CONFIG_NAME}"
OUTPUT "${OUTPUT_CONFIG_OPTION_NAME}"
OUTPUT "${OUTPUT_MAN_NAME}"
DEPENDS "${INPUT_CONFIG_NAME}"
DEPENDS CREATE_CONFIG_FOLDER
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py" "${INPUT_CONFIG_NAME}" "${OUTPUT_CONFIG_NAME}" "${OUTPUT_MAN_NAME}"
"${OUTPUT_CONFIG_OPTION_NAME}"
)
add_custom_command(
OUTPUT "${OUTPUT_MAN_NAME_COMPRESS}"
DEPENDS "${OUTPUT_MAN_NAME}"
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}"
)
set_source_files_properties(${OUTPUT_CONFIG_NAME} PROPERTIES
GENERATED TRUE)
set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
@@ -300,28 +244,29 @@ set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
set_source_files_properties(${OUTPUT_MAN_NAME} PROPERTIES
GENERATED TRUE)
set_source_files_properties(${OUTPUT_MAN_NAME_COMPRESS} PROPERTIES
GENERATED TRUE)
# Create the target
add_custom_target(CONFIG_INC
DEPENDS "${OUTPUT_CONFIG_NAME}"
DEPENDS "${OUTPUT_CONFIG_OPTION_NAME}"
DEPENDS "${OUTPUT_MAN_NAME}"
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
DEPENDS "${OUTPUT_MAN_NAME}")
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
# Install the man page
install(FILES ${OUTPUT_MAN_NAME} DESTINATION ${MAN_DIR}/man1)
# Add in diagnostic colours if the option is available.
# Ninja code generator will kill colours if this isn't here
check_cxx_compiler_flag(-fdiagnostics-color=always GCC_COLOR)
check_cxx_compiler_flag(-fcolor-diagnostics CLANG_COLOR)
function(AddDefaultOptionsToTarget Name)
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
set_target_properties(${Name} PROPERTIES VISIBILITY_INLINES_HIDDEN TRUE)
function(AddObject Name Type)
add_library(${Name} ${Type} ${SRCS})
add_dependencies(${Name} IR_INC)
add_dependencies(${Name} CONFIG_INC)
target_link_libraries(${Name} pthread rt vixl ${LINUX_LIBS} dl)
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
target_include_directories(${Name} PRIVATE IncludePrivate/)
@@ -331,18 +276,13 @@ function(AddDefaultOptionsToTarget Name)
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
target_compile_definitions(${Name} PRIVATE ${DEFINES})
add_dependencies(${Name} CONFIG_INC)
target_compile_options(${Name}
PRIVATE
-Wall
-Werror=cast-qual
-Werror=ignored-qualifiers
-Werror=implicit-fallthrough
-Wno-trigraphs
-ffunction-sections
-fwrapv
)
if (GCC_COLOR)
@@ -355,38 +295,16 @@ function(AddDefaultOptionsToTarget Name)
PRIVATE
"-fcolor-diagnostics")
endif()
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
target_link_options(${Name}
PRIVATE
"LINKER:--gc-sections"
"LINKER:--strip-all"
"LINKER:--as-needed"
)
endif()
endfunction()
# Build FEXCore_Config static library
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
target_link_libraries(FEXCore_Base ${LIBS})
AddDefaultOptionsToTarget(FEXCore_Base)
function(AddObject Name Type)
add_library(${Name} ${Type} ${SRCS})
add_dependencies(${Name} IR_INC)
target_link_libraries(${Name} FEXCore_Base)
AddDefaultOptionsToTarget(${Name})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
endfunction()
function(AddLibrary Name Type)
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
target_link_libraries(${Name} FEXCore_Base)
target_link_libraries(${Name} pthread rt vixl ${LINUX_LIBS} dl)
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
AddDefaultOptionsToTarget(${Name})
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
target_include_directories(${Name} PUBLIC "${PROJECT_SOURCE_DIR}/include/")
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
endfunction()
AddObject(${PROJECT_NAME}_object OBJECT)
+13 -18
View File
@@ -1,16 +1,12 @@
#pragma once
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/MathUtils.h>
#include "Common/MathUtils.h"
#include <FEXCore/Utils/LogManager.h>
#include <cstdint>
#include <cstdlib>
#include <cstring>
#include <stdint.h>
#include <stdlib.h>
#include <type_traits>
namespace FEXCore {
template<typename T>
struct BitSet final {
using ElementType = T;
@@ -20,16 +16,16 @@ struct BitSet final {
ElementType *Memory;
void Allocate(size_t Elements) {
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
LogMan::Throw::A((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(malloc(AllocateSize));
}
void Realloc(size_t Elements) {
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
LogMan::Throw::A((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(realloc(Memory, AllocateSize));
}
void Free() {
FEXCore::Allocator::free(Memory);
free(Memory);
Memory = nullptr;
}
bool Get(T Element) {
@@ -64,8 +60,8 @@ struct BitSetView final {
ElementType *Memory;
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0,
"Bitset view offset needs to be aligned to size of backing element");
LogMan::Throw::A((ElementOffset % MinimumSize) == 0,
"Bitset view offset needs to be aligned to size of backing element");
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
}
@@ -90,12 +86,11 @@ struct BitSetView final {
bool operator[](T Element) {
return Get(Element);
}
};
static_assert(sizeof(BitSet<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
static_assert(std::is_trivially_copyable_v<BitSet<uint32_t>>, "Needs to trivially copyable");
static_assert(std::is_trivially_copyable<BitSet<uint32_t>>::value, "Needs to trivially copyable");
static_assert(sizeof(BitSetView<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
static_assert(std::is_trivially_copyable_v<BitSetView<uint32_t>>, "Needs to trivially copyable");
} // namespace FEXCore
static_assert(std::is_trivially_copyable<BitSetView<uint32_t>>::value, "Needs to trivially copyable");
+22 -63
View File
@@ -1,86 +1,45 @@
#include "Common/JitSymbols.h"
#include <fcntl.h>
#include <string>
#include <sstream>
#include <unistd.h>
#include <fmt/format.h>
namespace FEXCore {
JITSymbols::JITSymbols() {
std::stringstream PerfMap;
PerfMap << "/tmp/perf-" << getpid() << ".map";
fp = fopen(PerfMap.str().c_str(), "wb");
if (fp) {
// Disable buffering on this file
setvbuf(fp, nullptr, _IONBF, 0);
}
}
JITSymbols::~JITSymbols() {
if (fd != -1) {
close(fd);
if (fp) {
fclose(fp);
}
}
void JITSymbols::InitFile() {
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
}
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (fd == -1) return;
void JITSymbols::Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fmt::format("{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
std::stringstream String;
String << std::hex << HostAddr << " " << CodeSize << " " << "JIT_0x" << GuestAddr << "_" << HostAddr << std::endl;
fwrite(String.str().c_str(), 1, String.str().size(), fp);
}
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
void JITSymbols::Register(void *HostAddr, uint32_t CodeSize, std::string const &Name) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fmt::format("{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
std::stringstream String;
String << std::hex << HostAddr << " " << CodeSize << " " << Name << "_" << HostAddr << std::endl;
fwrite(String.str().c_str(), 1, String.str().size(), fp);
}
}
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fmt::format("{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
} // namespace FEXCore
+4 -11
View File
@@ -1,24 +1,17 @@
#pragma once
#include <cstdint>
#include <cstdio>
#include <memory>
#include <string_view>
#include <string>
namespace FEXCore {
class JITSymbols final {
public:
JITSymbols();
~JITSymbols();
void InitFile();
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
void Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(void *HostAddr, uint32_t CodeSize, std::string const &Name);
private:
int fd{-1};
FILE* fp{};
};
}
+13
View File
@@ -0,0 +1,13 @@
#pragma once
#include <stdint.h>
static inline uint64_t AlignUp(uint64_t value, uint64_t size) {
return value + (size - value % size) % size;
};
static inline uint64_t AlignDown(uint64_t value, uint64_t size) {
return value - value % size;
};
@@ -1,47 +1,19 @@
#include <FEXCore/Utils/NetStream.h>
#include "NetStream.h"
#include <array>
#include <cstring>
#include <iterator>
#include <sys/types.h>
#include <sys/socket.h>
#include <stdio.h>
#include <unistd.h>
namespace FEXCore::Utils {
namespace {
class NetBuf final : public std::streambuf {
public:
explicit NetBuf(int socketfd) : socket{socketfd} {
reset_output_buffer();
}
~NetBuf() override {
close(socket);
}
private:
std::streamsize xsputn(const char* buffer, std::streamsize size) override;
std::streambuf::int_type underflow() override;
std::streambuf::int_type overflow(std::streambuf::int_type ch) override;
int sync() override;
void reset_output_buffer() {
// we always leave room for one extra char
setp(std::begin(output_buffer), std::end(output_buffer) -1);
}
int flushBuffer(const char *buffer, size_t size);
int socket;
std::array<char, 1400> output_buffer;
std::array<char, 1500> input_buffer; // enough for a typical packet
};
int NetBuf::flushBuffer(const char *buffer, size_t size) {
int NetStream::NetBuf::flushBuffer(const char *buffer, size_t size) {
size_t total = 0;
// Send data
while (total < size) {
size_t sent = send(socket, (const void*)(buffer + total), size - total, MSG_NOSIGNAL);
size_t sent = send(socket, (const void*)(buffer + total), size - total, 0);
if (sent == -1) {
// lets just assume all errors are end of file.
return -1;
@@ -52,12 +24,12 @@ int NetBuf::flushBuffer(const char *buffer, size_t size) {
return 0;
}
std::streamsize NetBuf::xsputn(const char* buffer, std::streamsize size) {
std::streamsize NetStream::NetBuf::xsputn(const char* buffer, std::streamsize size) {
size_t buf_remaining = epptr() - pptr();
// Check if the string fits neatly in our buffer
if (size <= buf_remaining) {
::memcpy(pptr(), buffer, size);
std::memcpy(pptr(), buffer, size);
pbump(size);
return size;
}
@@ -76,23 +48,23 @@ std::streamsize NetBuf::xsputn(const char* buffer, std::streamsize size) {
}
}
std::streambuf::int_type NetBuf::overflow(std::streambuf::int_type ch) {
std::streambuf::int_type NetStream::NetBuf::overflow(std::streambuf::int_type ch) {
// we always leave room for one extra char
*pptr() = (char) ch;
pbump(1);
return sync();
}
int NetBuf::sync() {
int NetStream::NetBuf::sync() {
// Flush and reset output buffer to zero
if (flushBuffer(pbase(), pptr() - pbase()) < 0) {
if(flushBuffer(pbase(), pptr() - pbase()) < 0) {
return -1;
}
reset_output_buffer();
return 0;
}
std::streambuf::int_type NetBuf::underflow() {
std::streambuf::int_type NetStream::NetBuf::underflow() {
ssize_t size = recv(socket, (void *)std::begin(input_buffer), sizeof(input_buffer), 0);
if (size <= 0) {
@@ -104,12 +76,11 @@ std::streambuf::int_type NetBuf::underflow() {
return traits_type::to_int_type(*gptr());
}
} // Anonymous namespace
NetStream::NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
NetStream::~NetStream() {
delete rdbuf();
}
} // namespace FEXCore::Utils
NetStream::NetBuf::~NetBuf() {
close(socket);
}
+41
View File
@@ -0,0 +1,41 @@
#pragma once
#include <array>
#include <iostream>
#include <string.h>
class NetStream : public std::iostream {
public:
NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
virtual ~NetStream();
private:
class NetBuf : public std::streambuf {
public:
NetBuf(int socketfd) {
socket = socketfd;
reset_output_buffer();
}
virtual ~NetBuf();
protected:
virtual std::streamsize xsputn(const char* buffer, std::streamsize size);
virtual std::streambuf::int_type underflow();
virtual std::streambuf::int_type overflow(std::streambuf::int_type ch);
virtual int sync();
private:
void reset_output_buffer() {
// we always leave room for one extra char
setp(std::begin(output_buffer), std::end(output_buffer) -1);
}
int flushBuffer(const char *buffer, size_t size);
int socket;
std::array<char, 1400> output_buffer;
std::array<char, 1500> input_buffer; // enough for a typical packet
};
};
+12 -53
View File
@@ -3,48 +3,13 @@
#include <cstdlib>
#include <filesystem>
#include <memory>
#include <pwd.h>
#include <system_error>
#include <unistd.h>
#include <sys/stat.h>
namespace FEXCore::Paths {
std::unique_ptr<std::string> CachePath;
std::unique_ptr<std::string> EntryCache;
char const* FindUserHomeThroughUID() {
auto passwd = getpwuid(geteuid());
if (passwd) {
return passwd->pw_dir;
}
return nullptr;
}
const char *GetHomeDirectory() {
char const *HomeDir = getenv("HOME");
// Try to get home directory from uid
if (!HomeDir) {
HomeDir = FindUserHomeThroughUID();
}
// try the PWD
if (!HomeDir) {
HomeDir = getenv("PWD");
}
// Still doesn't exit? You get local
if (!HomeDir) {
HomeDir = ".";
}
return HomeDir;
}
std::string CachePath;
std::string EntryCache;
void InitializePaths() {
CachePath = std::make_unique<std::string>();
EntryCache = std::make_unique<std::string>();
char const *HomeDir = getenv("HOME");
if (!HomeDir) {
@@ -57,35 +22,29 @@ namespace FEXCore::Paths {
char *XDGDataDir = getenv("XDG_DATA_DIR");
if (XDGDataDir) {
*CachePath = XDGDataDir;
CachePath = XDGDataDir;
}
else {
if (HomeDir) {
*CachePath = HomeDir;
CachePath = HomeDir;
}
}
*CachePath += "/.fex-emu/";
*EntryCache = *CachePath + "/EntryCache/";
CachePath += "/.fex-emu/";
EntryCache = CachePath + "/EntryCache/";
std::error_code ec{};
// Ensure the folder structure is created for our Data
if (!std::filesystem::exists(*EntryCache, ec) &&
!std::filesystem::create_directories(*EntryCache, ec)) {
LogMan::Msg::DFmt("Couldn't create EntryCache directory: '{}'", *EntryCache);
if (!std::filesystem::exists(EntryCache) &&
!std::filesystem::create_directories(EntryCache)) {
LogMan::Msg::D("Couldn't create EntryCache directory: '%s'", EntryCache.c_str());
}
}
void ShutdownPaths() {
CachePath.reset();
EntryCache.reset();
}
std::string GetCachePath() {
return *CachePath;
return CachePath;
}
std::string GetEntryCachePath() {
return *EntryCache;
return EntryCache;
}
}
-4
View File
@@ -3,10 +3,6 @@
namespace FEXCore::Paths {
void InitializePaths();
void ShutdownPaths();
const char *GetHomeDirectory();
std::string GetCachePath();
std::string GetEntryCachePath();
}
+27 -315
View File
@@ -1,6 +1,4 @@
#pragma once
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <cmath>
@@ -16,19 +14,9 @@ extern "C" {
struct X80SoftFloat {
#ifdef _M_X86_64
// Define this to push some operations to x87
// Only useful to see if precision loss is killing something
// #define DEBUG_X86_FLOAT
#ifdef DEBUG_X86_FLOAT
#define BIGFLOAT long double
#define BIGFLOATSIZE 10
#else
#define BIGFLOAT __float128
#define BIGFLOATSIZE 16
#endif
#elif defined(_M_ARM_64)
#define BIGFLOAT long double
#define BIGFLOATSIZE 16
#else
#error No 128bit float for this target!
#endif
@@ -57,183 +45,51 @@ struct X80SoftFloat {
// Ops
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
faddp;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_add(lhs, rhs);
#endif
}
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fsubp;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_sub(lhs, rhs);
#endif
}
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fmulp;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_mul(lhs, rhs);
#endif
}
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fdivp;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_div(lhs, rhs);
#endif
}
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fprem;
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
X80SoftFloat Rem = extF80_rem(lhs, rhs);
if (SignBit(Rem)) {
Rem = extF80_add(Rem, rhs);
}
else {
Rem.Sign = SignBit(lhs);
}
return Result;
#else
return extF80_rem(lhs, rhs);
#endif
return Rem;
}
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fprem1;
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
return extF80_rem(lhs, rhs);
#endif
}
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
}
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
return extF80_roundToInt(lhs, RoundMode, false);
}
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fxtract;
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st", "st(1)");
return Result;
#else
X80SoftFloat Tmp = lhs;
Tmp.Exponent = 0x3FFF;
Tmp.Sign = lhs.Sign;
return Tmp;
#endif
}
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fxtract;
ffreep %%st(0);
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st", "st(1)");
return Result;
#else
int32_t TrueExp = lhs.Exponent - ExponentBias;
return i32_to_extF80(TrueExp);
#endif
}
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
@@ -243,211 +99,77 @@ struct X80SoftFloat {
}
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fscale; # st0 = st0 * 2^(rdint(st1))
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
X80SoftFloat Int = FRNDINT(rhs, softfloat_round_minMag);
WARN_ONCE("x87: Application used FSCALE which may have accuracy problems");
X80SoftFloat Int = FRNDINT(rhs);
BIGFLOAT Src2_d = Int;
Src2_d = exp2l(Src2_d);
X80SoftFloat Src2_X80 = Src2_d;
X80SoftFloat Result = extF80_mul(lhs, Src2_X80);
return Result;
#endif
}
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
f2xm1; # st0 = 2^st(0) - 1
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
WARN_ONCE("x87: Application used F2XM1 which may have accuracy problems");
BIGFLOAT Src1_d = lhs;
BIGFLOAT Result = exp2l(Src1_d);
Result -= 1.0;
return Result;
#endif
}
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[rhs]; # st(1)
fldt %[lhs]; # st(0)
fyl2x; # st(1) * log2l(st(0))
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
WARN_ONCE("x87: Application used FYL2X which may have accuracy problems");
BIGFLOAT Src1_d = lhs;
BIGFLOAT Src2_d = rhs;
BIGFLOAT Tmp = Src2_d * log2l(Src1_d);
return Tmp;
#endif
}
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs];
fldt %[rhs];
fpatan;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
WARN_ONCE("x87: Application used FATAN which may have accuracy problems");
BIGFLOAT Src1_d = lhs;
BIGFLOAT Src2_d = rhs;
BIGFLOAT Tmp = atan2l(Src1_d, Src2_d);
return Tmp;
#endif
}
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fptan;
ffreep %%st(0);
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
WARN_ONCE("x87: Application used FTAN which may have accuracy problems");
BIGFLOAT Src_d = lhs;
Src_d = tanl(Src_d);
return Src_d;
#endif
}
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fsin;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
WARN_ONCE("x87: Application used FSIN which may have accuracy problems");
BIGFLOAT Src_d = lhs;
Src_d = sinl(Src_d);
return Src_d;
#endif
}
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fcos;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
WARN_ONCE("x87: Application used FCOS which may have accuracy problems");
BIGFLOAT Src_d = lhs;
Src_d = cosl(Src_d);
return Src_d;
#endif
}
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm (R"(
fninit;
fldt %[lhs]; # st0
fsqrt;
fstpt %[result];
)"
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
return extF80_sqrt(lhs);
#endif
}
operator float() const {
const float32_t Result = extF80_to_f32(*this);
return FEXCore::BitCast<float>(Result);
float32_t Result = extF80_to_f32(*this);
return *(float*)&Result;
}
operator double() const {
const float64_t Result = extF80_to_f64(*this);
return FEXCore::BitCast<double>(Result);
float64_t Result = extF80_to_f64(*this);
return *(double*)&Result;
}
operator BIGFLOAT() const {
#if BIGFLOATSIZE == 16
const float128_t Result = extF80_to_f128(*this);
return FEXCore::BitCast<BIGFLOAT>(Result);
#else
BIGFLOAT result{};
memcpy(&result, this, sizeof(result));
return result;
#endif
float128_t Result = extF80_to_f128(*this);
return *(BIGFLOAT*)&Result;
}
operator int16_t() const {
@@ -474,11 +196,11 @@ struct X80SoftFloat {
}
void operator=(const float rhs) {
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
*this = f32_to_extF80(*(float32_t*)&rhs);
}
void operator=(const double rhs) {
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
*this = f64_to_extF80(*(float64_t*)&rhs);
}
void operator=(const int16_t rhs) {
@@ -493,12 +215,6 @@ struct X80SoftFloat {
*this = ui64_to_extF80(rhs);
}
#if BIGFLOATSIZE == 10
void operator=(const long double rhs) {
memcpy(this, &rhs, sizeof(rhs));
}
#endif
operator void*() {
return reinterpret_cast<void*>(this);
}
@@ -510,19 +226,15 @@ struct X80SoftFloat {
}
X80SoftFloat(const float rhs) {
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
*this = f32_to_extF80(*(float32_t*)&rhs);
}
X80SoftFloat(const double rhs) {
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
*this = f64_to_extF80(*(float64_t*)&rhs);
}
X80SoftFloat(BIGFLOAT rhs) {
#if BIGFLOATSIZE == 16
*this = f128_to_extF80(FEXCore::BitCast<float128_t>(rhs));
#else
*this = FEXCore::BitCast<long double>(rhs);
#endif
*this = f128_to_extF80(*(float128_t*)&rhs);
}
X80SoftFloat(const int16_t rhs) {
-29
View File
@@ -1,29 +0,0 @@
#pragma once
#include <string>
namespace FEXCore::StringUtils {
// Trim the left side of the string of whitespace and new lines
[[maybe_unused]] static std::string LeftTrim(std::string String, std::string TrimTokens = " \t\n\r") {
size_t pos = std::string::npos;
if ((pos = String.find_first_not_of(TrimTokens)) != std::string::npos) {
String.erase(0, pos);
}
return String;
}
// Trim the right side of the string of whitespace and new lines
[[maybe_unused]] static std::string RightTrim(std::string String, std::string TrimTokens = " \t\n\r") {
size_t pos = std::string::npos;
if ((pos = String.find_last_not_of(TrimTokens)) != std::string::npos) {
String.erase(String.begin() + pos + 1, String.end());
}
return String;
}
// Trim both the left and right of the string of whitespace and new lines
[[maybe_unused]] static std::string Trim(std::string String, std::string TrimTokens = " \t\n\r") {
return RightTrim(LeftTrim(String, TrimTokens), TrimTokens);
}
}
+59 -435
View File
@@ -1,127 +1,43 @@
#include "Common/StringConv.h"
#include "Common/StringUtils.h"
#include "Common/Paths.h"
#include "Utils/FileLoading.h"
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Context/Context.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
#include <assert.h>
#include <cstdlib>
#include <filesystem>
#include <fstream>
#include <functional>
#include <pwd.h>
#include <map>
#include <memory>
#include <list>
#include <optional>
#include <stddef.h>
#include <stdint.h>
#include <string>
#include <string_view>
#include <sys/sysinfo.h>
#include <system_error>
#include <type_traits>
#include <unordered_map>
#include <utility>
#include <vector>
#include <tiny-json.h>
namespace FEXCore::Context {
struct Context;
}
#include <unistd.h>
namespace FEXCore::Config {
namespace DefaultValues {
#define P(x) x
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
#include <FEXCore/Config/ConfigValues.inl>
}
namespace JSON {
struct JsonAllocator {
jsonPool_t PoolObject;
std::unique_ptr<std::list<json_t>> json_objects;
};
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
json_t* PoolInit(jsonPool_t* Pool) {
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
alloc->json_objects = std::make_unique<std::list<json_t>>();
return &*alloc->json_objects->emplace(alloc->json_objects->end());
char const* FindUserHomeThroughUID() {
auto passwd = getpwuid(geteuid());
if (passwd) {
return passwd->pw_dir;
}
return nullptr;
}
json_t* PoolAlloc(jsonPool_t* Pool) {
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
return &*alloc->json_objects->emplace(alloc->json_objects->end());
}
const char *GetHomeDirectory() {
char const *HomeDir = getenv("HOME");
static void LoadJSonConfig(const std::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
std::vector<char> Data;
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
return;
// Try to get home directory from uid
if (!HomeDir) {
HomeDir = FindUserHomeThroughUID();
}
JsonAllocator Pool {
.PoolObject = {
.init = PoolInit,
.alloc = PoolAlloc,
},
};
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
if (!json) {
LogMan::Msg::EFmt("Couldn't create json");
return;
// try the PWD
if (!HomeDir) {
HomeDir = getenv("PWD");
}
json_t const* ConfigList = json_getProperty(json, "Config");
if (!ConfigList) {
// This is a non-error if the configuration file exists but no Config section
return;
// Still doesn't exit? You get local
if (!HomeDir) {
HomeDir = ".";
}
for (json_t const* ConfigItem = json_getChild(ConfigList);
ConfigItem != nullptr;
ConfigItem = json_getSibling(ConfigItem)) {
const char* ConfigName = json_getName(ConfigItem);
const char* ConfigString = json_getValue(ConfigItem);
if (!ConfigName) {
LogMan::Msg::EFmt("Couldn't get config name");
return;
}
if (!ConfigString) {
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
return;
}
Func(ConfigName, ConfigString);
}
}
}
std::string GetDataDirectory() {
std::string DataDir{};
char const *HomeDir = Paths::GetHomeDirectory();
char const *DataXDG = getenv("XDG_DATA_HOME");
char const *DataOverride = getenv("FEX_APP_DATA_LOCATION");
if (DataOverride) {
// Data override will override the complete directory
DataDir = DataOverride;
}
else {
DataDir = DataXDG ?: HomeDir;
DataDir += "/.fex-emu/";
}
return DataDir;
return HomeDir;
}
std::string GetConfigDirectory(bool Global) {
@@ -130,22 +46,15 @@ namespace JSON {
ConfigDir = GLOBAL_DATA_DIRECTORY;
}
else {
char const *HomeDir = Paths::GetHomeDirectory();
char const *HomeDir = GetHomeDirectory();
char const *ConfigXDG = getenv("XDG_CONFIG_HOME");
char const *ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
if (ConfigOverride) {
// Config override completely overrides the config directory
ConfigDir = ConfigOverride;
}
else {
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
ConfigDir += "/.fex-emu/";
}
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
ConfigDir += "/.fex-emu/";
// Ensure the folder structure is created for our configuration
std::error_code ec{};
if (!std::filesystem::exists(ConfigDir, ec) &&
!std::filesystem::create_directories(ConfigDir, ec)) {
if (!std::filesystem::exists(ConfigDir) &&
!std::filesystem::create_directories(ConfigDir)) {
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigDir.c_str());
// Let's go local in this case
return "./";
}
@@ -154,50 +63,35 @@ namespace JSON {
return ConfigDir;
}
std::string GetConfigFileLocation(bool Global) {
std::string ConfigFile{};
if (Global) {
ConfigFile = GetConfigDirectory(true) + "Config.json";
}
else {
const char *AppConfig = getenv("FEX_APP_CONFIG");
if (AppConfig) {
// App config environment variable overwrites only the config file
ConfigFile = AppConfig;
}
else {
ConfigFile = GetConfigDirectory(false) + "Config.json";
}
}
std::string GetConfigFileLocation() {
std::string ConfigFile = GetConfigDirectory(false) + "Config.json";
return ConfigFile;
}
std::string GetApplicationConfig(const std::string &Filename, bool Global) {
std::string GetApplicationConfig(std::string &Filename, bool Global) {
std::string ConfigFile = GetConfigDirectory(Global);
std::error_code ec{};
if (!Global &&
!std::filesystem::exists(ConfigFile, ec) &&
!std::filesystem::create_directories(ConfigFile, ec)) {
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
!std::filesystem::exists(ConfigFile) &&
!std::filesystem::create_directories(ConfigFile)) {
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigFile.c_str());
// Let's go local in this case
return "./" + Filename + ".json";
return "./";
}
ConfigFile += "AppConfig/";
// Attempt to create the local folder if it doesn't exist
if (!Global &&
!std::filesystem::exists(ConfigFile, ec) &&
!std::filesystem::create_directories(ConfigFile, ec)) {
// Let's go local in this case
return "./" + Filename + ".json";
}
ConfigFile += Filename + ".json";
ConfigFile += "AppConfig/" + Filename + ".json";
return ConfigFile;
}
std::string GetDataDirectory() {
std::string DataDir{};
char const *HomeDir = GetHomeDirectory();
char const *DataXDG = getenv("XDG_DATA_HOME");
DataDir = DataXDG ?: HomeDir;
DataDir += "/.fex-emu/";
return DataDir;
}
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
}
@@ -211,12 +105,9 @@ namespace JSON {
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
static FEXCore::Config::Layer *Meta{};
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
constexpr std::array<FEXCore::Config::LayerType, 6> LoadOrder = {
FEXCore::Config::LayerType::LAYER_MAIN,
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP,
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
FEXCore::Config::LayerType::LAYER_ARGUMENTS,
FEXCore::Config::LayerType::LAYER_ENVIRONMENT,
@@ -300,8 +191,7 @@ namespace JSON {
void MetaLayer::MergeConfigMap(const LayerOptions &Options) {
// Insert this layer's options, overlaying previous options that exist here
for (auto &it : Options) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV ||
it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV) {
MergeEnvironmentVariables(it.first, it.second);
}
else {
@@ -329,7 +219,7 @@ namespace JSON {
}
}
std::string ExpandPath(std::string const &ContainerPrefix, std::string PathName) {
std::string ExpandPath(std::string PathName) {
if (PathName.empty()) {
return {};
}
@@ -350,67 +240,10 @@ namespace JSON {
Path = std::filesystem::absolute(Path);
// Only return if it exists
std::error_code ec{};
if (std::filesystem::exists(Path, ec)) {
if (std::filesystem::exists(Path)) {
return Path;
}
}
else {
// If the containerprefix and pathname isn't empty
// Then we check if the pathname exists in our current namespace
// If the path DOESN'T exist but DOES exist with the prefix applied
// then redirect to the prefix
//
// This might not be expected behaviour for some edge cases but since
// all paths aren't mounted inside the container, then it'll be fine
//
// Main catch case for this is the default thunk install folders
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
if (!ContainerPrefix.empty() && !PathName.empty()) {
if (!std::filesystem::exists(PathName)) {
auto ContainerPath = ContainerPrefix + PathName;
if (std::filesystem::exists(ContainerPath)) {
return ContainerPath;
}
}
}
}
return {};
}
std::string FindContainer() {
// We only support pressure-vessel at the moment
const static std::string ContainerManager = "/run/host/container-manager";
if (std::filesystem::exists(ContainerManager)) {
std::vector<char> Manager{};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
std::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
return ManagerStr;
}
}
return {};
}
std::string FindContainerPrefix() {
// We only support pressure-vessel at the moment
const static std::string ContainerManager = "/run/host/container-manager";
if (std::filesystem::exists(ContainerManager)) {
std::vector<char> Manager{};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
std::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
// We are running inside of pressure vessel
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
return "/run/host/";
}
}
}
return {};
}
@@ -418,9 +251,7 @@ namespace JSON {
Meta->Load();
// Do configuration option fix ups after everything is reloaded
{
// Always fix up the number of threads and create the configuration
// Otherwise the application could receive zero as the number of threads
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THREADS)) {
FEX_CONFIG_OPT(Cores, THREADS);
if (Cores == 0) {
// When the number of emulated CPU cores is zero then auto detect
@@ -428,38 +259,8 @@ namespace JSON {
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
// Sanitize Core option
FEX_CONFIG_OPT(Core, CORE);
#if (_M_X86_64)
constexpr uint32_t MaxCoreNumber = 2;
#else
constexpr uint32_t MaxCoreNumber = 1;
#endif
#ifdef INTERPRETER_ENABLED
constexpr uint32_t MinCoreNumber = 0;
#else
constexpr uint32_t MinCoreNumber = 1;
#endif
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
// Sanitize the core option by setting the core to the JIT if invalid
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, std::to_string(FEXCore::Config::CONFIG_IRJIT));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(Core, CORE);
if (CacheObjectCodeCompilation() && Core() == FEXCore::Config::CONFIG_INTERPRETER) {
// If running the interpreter then disable cache code compilation
FEXCore::Config::Erase(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION);
}
}
std::string ContainerPrefix { FindContainerPrefix() };
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
auto NewPath = ExpandPath(ContainerPrefix, PathName);
auto ExpandPathIfExists = [](FEXCore::Config::ConfigOption Config, std::string PathName) {
auto NewPath = ExpandPath(PathName);
if (!NewPath.empty()) {
FEXCore::Config::EraseSet(Config, NewPath);
}
@@ -467,7 +268,7 @@ namespace JSON {
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
FEX_CONFIG_OPT(PathName, ROOTFS);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
auto ExpandedString = ExpandPath(PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
@@ -475,8 +276,7 @@ namespace JSON {
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
std::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
std::error_code ec{};
if (std::filesystem::exists(NamedRootFS, ec)) {
if (std::filesystem::exists(NamedRootFS)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
}
}
@@ -491,19 +291,7 @@ namespace JSON {
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
}
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
std::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
std::error_code ec{};
if (std::filesystem::exists(NamedConfig, ec)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
}
}
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKCONFIG, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
@@ -534,15 +322,11 @@ namespace JSON {
return Meta->Get(Option);
}
void Set(ConfigOption Option, std::string_view Data) {
void Set(ConfigOption Option, std::string Data) {
Meta->Set(Option, Data);
}
void Erase(ConfigOption Option) {
Meta->Erase(Option);
}
void EraseSet(ConfigOption Option, std::string_view Data) {
void EraseSet(ConfigOption Option, std::string Data) {
Meta->EraseSet(Option, Data);
}
@@ -581,17 +365,6 @@ namespace JSON {
}
}
template<>
std::string Value<std::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
}
else {
return std::string(Default);
}
}
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
@@ -616,154 +389,5 @@ namespace JSON {
*List = **Value;
}
}
template void Value<std::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, std::list<std::string> *List);
// Application loaders
class MainLoader final : public FEXCore::Config::OptionMapper {
public:
explicit MainLoader(FEXCore::Config::LayerType Type);
explicit MainLoader(std::string ConfigFile);
void Load() override;
private:
std::string Config;
};
class AppLoader final : public FEXCore::Config::OptionMapper {
public:
explicit AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type);
void Load();
private:
std::string Config;
};
class EnvLoader final : public FEXCore::Config::Layer {
public:
explicit EnvLoader(char *const _envp[]);
void Load() override;
private:
char *const *envp;
};
static const std::map<std::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
#include <FEXCore/Config/ConfigValues.inl>
}};
static const std::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
#include <FEXCore/Config/ConfigValues.inl>
}};
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
: FEXCore::Config::Layer(Layer) {
}
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
auto it = ConfigLookup.find(ConfigName);
if (it != ConfigLookup.end()) {
Set(it->second, ConfigString);
}
}
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
: FEXCore::Config::OptionMapper(Type)
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
}
MainLoader::MainLoader(std::string ConfigFile)
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
, Config{std::move(ConfigFile)} {
}
void MainLoader::Load() {
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
MapNameToOption(Name, ConfigString);
});
}
AppLoader::AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type)
: FEXCore::Config::OptionMapper(Type) {
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
// Immediately load so we can reload the meta layer
Load();
}
void AppLoader::Load() {
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
MapNameToOption(Name, ConfigString);
});
}
EnvLoader::EnvLoader(char *const _envp[])
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
, envp {_envp} {
}
void EnvLoader::Load() {
std::unordered_map<std::string_view, std::string_view> EnvMap;
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
std::string_view Var(*pvar);
size_t pos = Var.rfind('=');
if (std::string::npos == pos)
continue;
std::string_view Key = Var.substr(0,pos);
std::string_view Value {Var.substr(pos+1)};
#define ENVLOADER
#include <FEXCore/Config/ConfigOptions.inl>
EnvMap[Key]=Value;
}
std::function GetVar = [=](const std::string_view id) -> std::optional<std::string_view> {
if (EnvMap.find(id) != EnvMap.end())
return EnvMap.at(id);
// If envp[] was empty, search using std::getenv()
const char* vs = std::getenv(id.data());
if (vs) {
return vs;
}
else {
return std::nullopt;
}
};
std::optional<std::string_view> Value;
for (auto &it : EnvConfigLookup) {
if ((Value = GetVar(it.first)).has_value()) {
Set(it.second, std::string(*Value));
}
}
}
std::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
}
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
if (File) {
return std::make_unique<FEXCore::Config::MainLoader>(*File);
}
else {
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
}
}
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, FEXCore::Config::LayerType Type) {
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
}
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
return std::make_unique<FEXCore::Config::EnvLoader>(_envp);
}
}
+232
View File
@@ -0,0 +1,232 @@
{
"Options": {
"CPU": {
"Core": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
"TextDefault": "irjit",
"ShortArg": "c",
"Choices": [ "irint", "irjit", "host" ],
"ArgumentHandler": "CoreHandler",
"Desc": [
"Which CPU core to use",
"host only exists on x86_64",
"[irint, irjit, host]"
]
},
"Multiblock": {
"Type": "bool",
"Default": "true",
"ShortArg": "m",
"Desc": [
"Controls multiblock code compilation"
]
},
"MaxInst": {
"Type": "int32",
"Default": "5000",
"ShortArg": "n",
"Desc": [
"Maximum number of instruction to store in a block"
]
},
"Threads": {
"Type": "uint32",
"Default": "1",
"ShortArg": "T",
"Desc": [
"Number of physical hardware threads to tell the process we have.",
"0 will auto detect."
]
}
},
"Emulation": {
"RootFS": {
"Type": "str",
"Default": "",
"ShortArg": "R",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
"\teg: ~/RootFS/Debian_x86_64",
"Or this can be a name of a rootfs",
"If the named rootfs exists in the FEX data folder then it will use that one",
"\teg: $HOME/.fex-emu/RootFS/<RootFS name>/",
"Or if you have XDG_DATA_HOME the config will search in that directory",
"\teg: $XDG_DATA_HOME/.fex-emu/RootFS/<RootFS name>/"
]
},
"ThunkHostLibs": {
"Type": "str",
"Default": "",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
]
},
"ThunkGuestLibs": {
"Type": "str",
"Default": "",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
},
"ThunkConfig": {
"Type": "str",
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks."
]
},
"Env": {
"Type": "strarray",
"Default": "",
"ShortArg": "E",
"Desc": [
"Adds an environment variable to the emulated environment."
]
}
},
"Debug": {
"SingleStep": {
"Type": "bool",
"Default": "false",
"ShortArg": "S",
"Desc": [
"Single stepping configuration."
]
},
"GdbServer": {
"Type": "bool",
"Default": "false",
"ShortArg": "G",
"Desc": [
"Enables the GDB server."
]
},
"DumpIR": {
"Type": "str",
"Default": "no",
"Desc": [
"Folder to dump the IR in to.",
"[no, stdout, stderr, <Folder>]"
]
},
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"ShortArg": "g",
"Desc": [
"When the test harness ends, print the GPR state."
]
},
"O0": {
"Type": "bool",
"Default": "false",
"ShortArg": "O0",
"Desc": [
"Disables optimizations passes for debugging."
]
}
},
"Logging": {
"SilentLog": {
"Type": "bool",
"Default": "true",
"ShortArg": "s",
"Desc": [
"Disables logging"
]
},
"OutputLog": {
"Type": "str",
"Default": "stdout",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
"[stdout, stderr, <Filename>]"
]
}
},
"Hacks": {
"SMCChecks": {
"Type": "uint8",
"Default": "FEXCore::Config::CONFIG_SMC_MMAN",
"TextDefault": "mman",
"ArgumentHandler": "SMCCheckHandler",
"Desc": [
"Checks code for modification before execution.",
"\tnone: No checks",
"\tmman: Invalidate on mmap, mprotect, munmap",
"\tfull: Validate code before every run (slow)"
]
},
"TSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"Controls TSO IR ops.",
"Highly likely to break any multithreaded application if disabled."
]
},
"ABILocalFlags": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around flags.",
"Assumes flags are not used across cals.",
"Hand-written assembly can violate this assumption."
]
},
"ABINoPF": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around parity flag calculation.",
"Removes the calculation of the parity flag from GPR instructions.",
"Assuming no uses rely on it"
]
}
},
"Misc": {
"AOTIRCapture": {
"Type": "bool",
"Default": "false",
"Desc": [
"Captures IR and generates an AOT IR cache.",
"Captures both the loaded executable and libraries it loads."
]
},
"AOTIRLoad": {
"Type": "bool",
"Default": "false",
"Desc": [
"Loads an AOT IR cache for the loaded executable."
]
}
}
},
"UnnamedOptions": {
"Misc": {
"IS_INTERPRETER": {
"Type": "bool",
"Default": "false"
},
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false"
},
"APP_FILENAME": {
"Type": "str",
"Default": ""
},
"IS64BIT_MODE": {
"Type": "bool",
"Default": "false"
}
}
}
}
-411
View File
@@ -1,411 +0,0 @@
{
"Options": {
"CPU": {
"Core": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
"TextDefault": "irjit",
"ShortArg": "c",
"Choices": [ "irint", "irjit", "host" ],
"ArgumentHandler": "CoreHandler",
"Desc": [
"Which CPU core to use",
"host only exists on x86_64",
"[irint, irjit, host]"
]
},
"Multiblock": {
"Type": "bool",
"Default": "false",
"ShortArg": "m",
"Desc": [
"Controls multiblock code compilation",
"Can cause long JIT compilation times and stutter"
]
},
"MaxInst": {
"Type": "int32",
"Default": "5000",
"ShortArg": "n",
"Desc": [
"Maximum number of instruction to store in a block"
]
},
"Threads": {
"Type": "uint32",
"Default": "0",
"ShortArg": "T",
"Desc": [
"Number of physical hardware threads to tell the process we have.",
"0 will auto detect."
]
},
"CacheObjectCodeCompilation": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
"TextDefault": "none",
"Choices": [ "none", "read", "readwrite" ],
"ArgumentHandler": "CacheObjectCodeHandler",
"Desc": [
"Cache JIT object code to drive.",
"Allows JIT code to be shared between applications"
]
},
"EnableAVX": {
"Type": "bool",
"Default": "true",
"Desc": [
"Determines whether or not we use the expanded register file for AVX or not"
]
}
},
"Emulation": {
"RootFS": {
"Type": "str",
"Default": "",
"ShortArg": "R",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
"\teg: ~/RootFS/Debian_x86_64",
"Or this can be a name of a rootfs",
"If the named rootfs exists in the FEX data folder then it will use that one",
"\teg: $HOME/.fex-emu/RootFS/<RootFS name>/",
"Or if you have XDG_DATA_HOME the config will search in that directory",
"\teg: $XDG_DATA_HOME/.fex-emu/RootFS/<RootFS name>/"
]
},
"ThunkHostLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks/",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
]
},
"ThunkGuestLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks/",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
},
"ThunkHostLibs32": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks_32/",
"Desc": [
"Folder to find the 32-bit host-side thunking libraries."
]
},
"ThunkGuestLibs32": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks_32/",
"Desc": [
"Folder to find the 32-bit guest-side thunking libraries."
]
},
"ThunkConfig": {
"Type": "str",
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks.",
"This can be a filesystem path",
"\teg: ~/MyThunkConfig.json",
"Or this can be a named of a Thunk config file",
"If the named config file exists in the FEX data folder folder the it will use that one",
"\teg: $HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>",
"Or if you have XDG_DATA_HOME the config will search in that directory",
"\teg: $XDG_DATA_HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>"
]
},
"Env": {
"Type": "strarray",
"Default": "",
"ShortArg": "E",
"Desc": [
"Adds an environment variable to the emulated environment."
]
},
"HostEnv": {
"Type": "strarray",
"Default": "",
"ShortArg": "H",
"Desc": [
"Adds an environment variable to the host environment.",
"This can be useful for setting environment variables that thunks can pick up.",
"Typically isn't necessary since the guest libc isn't thunked. But is possible."
]
},
"AdditionalArguments": {
"Type": "strarray",
"Default": "",
"Desc": [
"Allows the user to pass additional arguments to the application"
]
}
},
"Debug": {
"SingleStep": {
"Type": "bool",
"Default": "false",
"ShortArg": "S",
"Desc": [
"Single stepping configuration."
]
},
"GdbServer": {
"Type": "bool",
"Default": "false",
"ShortArg": "G",
"Desc": [
"Enables the GDB server."
]
},
"DumpIR": {
"Type": "str",
"Default": "no",
"Desc": [
"Folder to dump the IR in to.",
"[no, stdout, stderr, <Folder>]"
]
},
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"ShortArg": "g",
"Desc": [
"When the test harness ends, print the GPR state."
]
},
"O0": {
"Type": "bool",
"Default": "false",
"ShortArg": "O0",
"Desc": [
"Disables optimizations passes for debugging."
]
},
"SRA": {
"Type": "bool",
"Default": "true",
"Desc": [
"Set to false to disable Static Register Allocation"
]
},
"Force32BitAllocator": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces use of the 32-bit allocator on 32-bit applications",
"Used to work around ulimit problems of CI runner",
"Potentially useful for debugging memory problems",
"32-bit allocator is always used if your host kernel is older than 4.17"
]
},
"GlobalJITNaming": {
"Type": "bool",
"Default": "false",
"Desc": [
"Uses JITSymbols to name all JIT state as one symbol",
"Useful for querying how much time is spent inside of the JIT",
"Profiling tools will show JIT time as FEXJIT"
]
},
"LibraryJITNaming": {
"Type": "bool",
"Default": "false",
"Desc": [
"Uses JITSymbols to name JIT symbols grouped by library",
"Useful for querying how much time is spent in each guest library",
"Can be used to help guide thunk generation"
]
},
"BlockJITNaming": {
"Type": "bool",
"Default": "false",
"Desc": [
"Uses JITSymbols to name JIT symbols",
"Useful for determining hot blocks of code",
"Has some file writing overhead per JIT block"
]
},
"GDBSymbols": {
"Type": "bool",
"Default": "false",
"Desc": [
"Integrates with GDB using the JIT interface.",
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
"Also needs x86_64-linux-gnu-objdump in PATH.",
"Can be very slow."
]
}
},
"Logging": {
"SilentLog": {
"Type": "bool",
"Default": "true",
"ShortArg": "s",
"Desc": [
"Disables logging"
]
},
"OutputLog": {
"Type": "str",
"Default": "server",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
"[stdout, stderr, server, <Filename>]"
]
}
},
"Hacks": {
"SMCChecks": {
"Type": "uint8",
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
"TextDefault": "mtrack",
"ArgumentHandler": "SMCCheckHandler",
"Desc": [
"Checks code for modification before execution.",
"\tnone: No checks",
"\tmtrack: Page tracking based invalidation",
"\tfull: Validate code before every run (slow)",
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
]
},
"TSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"Controls TSO IR ops.",
"Highly likely to break any multithreaded application if disabled."
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
"Desc": [
"Automatically enables TSO when shared memory is used.",
"Should work without issues in most cases."
]
},
"X87ReducedPrecision": {
"Type": "bool",
"Default": "false",
"Desc": [
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
]
},
"ABILocalFlags": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around flags.",
"Assumes flags are not used across cals.",
"Hand-written assembly can violate this assumption."
]
},
"ABINoPF": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around parity flag calculation.",
"Removes the calculation of the parity flag from GPR instructions.",
"Assuming no uses rely on it"
]
},
"ParanoidTSO": {
"Type": "bool",
"Default": "false",
"Desc": [
"Makes TSO operations even more strict.",
"Forces vector loadstores to also become atomic."
]
},
"StallProcess": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces a process to stall out on initialization",
"Useful for a process that keeps restarting and doesn't work"
]
},
"x86dec_SynchronizeRIPOnAllBlocks": {
"Type": "bool",
"Default": "false",
"Desc": [
"An application that uses try-catch or longjump extensively needs the ability to do context aware state flushing",
"In the case of FEX's block-linking, it won't always ensure that RIP is synchronized.",
"If an exception occurs and RIP isn't synchronized, then FEX's exception stack restore may not long jump as expected",
"Can be useful for Wine applications that rely on stack unwinding"
]
}
},
"Misc": {
"AOTIRCapture": {
"Type": "bool",
"Default": "false",
"Desc": [
"Captures IR and generates an AOT IR cache.",
"Captures both the loaded executable and libraries it loads."
]
},
"AOTIRGenerate": {
"Type": "bool",
"Default": "false",
"Desc": [
"Scans file for executable code and generates an AOT IR cache.",
"Does not run the executable."
]
},
"AOTIRLoad": {
"Type": "bool",
"Default": "false",
"Desc": [
"Loads an AOT IR cache for the loaded executable."
]
},
"ServerSocketPath": {
"Type": "str",
"Default": "",
"Desc": [
"Override for a FEXServer socket path. Only useful for chroots."
]
}
}
},
"UnnamedOptions": {
"Misc": {
"IS_INTERPRETER": {
"Type": "bool",
"Default": "false"
},
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false"
},
"APP_FILENAME": {
"Type": "str",
"Default": ""
},
"APP_CONFIG_NAME": {
"Type": "str",
"Default": "",
"Desc": [
"This is the application config name that has been loaded.",
"This differs from APP_FILENAME in two ways",
"Where APP_FILENAME always points to the executable path that FEX-Emu is executing.",
"This matches what is used to load the AppLayer configuration name.",
"When running through a compatibility layer like wine, this will only be the exe name, instead of wine full path."
]
},
"IS64BIT_MODE": {
"Type": "bool",
"Default": "false"
}
}
}
}
+28 -77
View File
@@ -2,20 +2,10 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/Core.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Core/SignalDelegator.h>
#include "FEXCore/Debug/InternalThreadState.h"
#include <string.h>
#include <utility>
namespace FEXCore::HLE {
class SyscallVisitor;
}
#include <FEXCore/Debug/X86Tables.h>
namespace FEXCore::Context {
void InitializeStaticTables(OperatingMode Mode) {
@@ -24,10 +14,6 @@ namespace FEXCore::Context {
IR::InstallOpcodeHandlers(Mode);
}
void ShutdownStaticTables() {
FEXCore::Paths::ShutdownPaths();
}
FEXCore::Context::Context *CreateNewContext() {
return new FEXCore::Context::Context{};
}
@@ -43,15 +29,16 @@ namespace FEXCore::Context {
delete CTX;
}
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, uint64_t InitialRIP, uint64_t StackPointer) {
return CTX->InitCore(InitialRIP, StackPointer);
bool InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
return CTX->InitCore(Loader);
}
void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler) {
CTX->CustomExitHandler = std::move(handler);
void SetExitHandler(FEXCore::Context::Context *CTX,
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> handler) {
CTX->CustomExitHandler = handler;
}
ExitHandler GetExitHandler(const FEXCore::Context::Context *CTX) {
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> GetExitHandler(FEXCore::Context::Context *CTX) {
return CTX->CustomExitHandler;
}
@@ -63,31 +50,28 @@ namespace FEXCore::Context {
CTX->Step();
}
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
Thread->CTX->CompileBlock(Thread->CurrentFrame, GuestRIP);
}
FEXCore::Context::ExitReason RunUntilExit(FEXCore::Context::Context *CTX) {
return CTX->RunUntilExit();
}
int GetProgramStatus(const FEXCore::Context::Context *CTX) {
int GetProgramStatus(FEXCore::Context::Context *CTX) {
return CTX->GetProgramStatus();
}
FEXCore::Context::ExitReason GetExitReason(const FEXCore::Context::Context *CTX) {
FEXCore::Context::ExitReason GetExitReason(FEXCore::Context::Context *CTX) {
return CTX->ParentThread->ExitReason;
}
bool IsDone(const FEXCore::Context::Context *CTX) {
bool IsDone(FEXCore::Context::Context *CTX) {
return CTX->IsPaused();
}
void GetCPUState(const FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
void GetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
memcpy(State, CTX->ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
}
void SetCPUState(FEXCore::Context::Context *CTX, const FEXCore::Core::CPUState *State) {
void SetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
memcpy(CTX->ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
}
@@ -110,30 +94,22 @@ namespace FEXCore::Context {
void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t Syscall, [[maybe_unused]] FEXCore::HLE::SyscallVisitor *Visitor) {
}
HostFeatures GetHostFeatures(const FEXCore::Context::Context *CTX) {
return CTX->HostFeatures;
void HandleCallback(FEXCore::Context::Context *CTX, uint64_t RIP) {
CTX->HandleCallback(RIP);
}
void HandleCallback(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
CTX->HandleCallback(Thread, RIP);
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
CTX->RegisterHostSignalHandler(Signal, Func);
}
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
CTX->RegisterHostSignalHandler(Signal, std::move(Func), Required);
}
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
CTX->RegisterFrontendHostSignalHandler(Signal, std::move(Func), Required);
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
CTX->RegisterFrontendHostSignalHandler(Signal, Func);
}
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
return CTX->CreateThread(NewThreadState, ParentTID);
}
void ExecutionThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
return CTX->ExecutionThread(Thread);
}
void InitializeThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
return CTX->InitializeThread(Thread);
}
@@ -153,57 +129,32 @@ namespace FEXCore::Context {
void CleanupAfterFork(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
CTX->CleanupAfterFork(Thread);
}
void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation) {
CTX->SignalDelegation = SignalDelegation;
}
void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler) {
CTX->SyscallHandler = Handler;
CTX->SourcecodeResolver = Handler->GetSourcecodeResolver();
}
FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf) {
return CTX->CPUID.RunFunction(Function, Leaf);
}
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf, uint32_t CPU) {
return CTX->CPUID.RunFunctionName(Function, Leaf, CPU);
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::istream>(const std::string&)> CacheReader) {
CTX->AOTIRLoader = CacheReader;
}
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader) {
CTX->SetAOTIRLoader(CacheReader);
bool WriteAOTIR(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
return CTX->WriteAOTIRCache(CacheWriter);
}
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
CTX->SetAOTIRWriter(CacheWriter);
void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name) {
return CTX->AddNamedRegion(Base, Length, Offset, Name);
}
void SetAOTIRRenamer(FEXCore::Context::Context *CTX, std::function<void(const std::string&)> CacheRenamer) {
CTX->SetAOTIRRenamer(CacheRenamer);
}
void FinalizeAOTIRCache(FEXCore::Context::Context *CTX) {
CTX->FinalizeAOTIRCache();
}
void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
CTX->WriteFilesWithCode(Writer);
}
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(FEXCore::Context::Context *CTX, const std::string &Name) {
return CTX->LoadAOTIRCacheEntry(Name);
}
void UnloadAOTIRCacheEntry(FEXCore::Context::Context *CTX, IR::AOTIRCacheEntry *Entry) {
return CTX->UnloadAOTIRCacheEntry(Entry);
}
CustomIRResult AddCustomIREntrypoint(FEXCore::Context::Context *CTX, uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
return CTX->AddCustomIREntrypoint(Entrypoint, Handler, Creator, Data);
}
void AppendThunkDefinitions(FEXCore::Context::Context *CTX, std::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
CTX->AppendThunkDefinitions(Definitions);
void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length) {
return CTX->RemoveNamedRegion(Base, Length);
}
namespace Debug {
+77 -216
View File
@@ -1,62 +1,44 @@
#pragma once
#include "Common/JitSymbols.h"
#include "FEXHeaderUtils/ScopedSignalMask.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/HostFeatures.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/AOTIR.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Utils/Event.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <stdint.h>
#include <atomic>
#include <condition_variable>
#include <functional>
#include <istream>
#include <map>
#include <memory>
#include <mutex>
#include <shared_mutex>
#include <stddef.h>
#include <string>
#include <optional>
#include <ostream>
#include <set>
#include <unordered_map>
#include <queue>
#include <vector>
namespace FEXCore {
class CodeLoader;
class ThunkHandler;
class BlockSamplingData;
class GdbServer;
namespace CodeSerialize {
class CodeObjectSerializeService;
}
class SiganlDelegator;
namespace CPU {
class Arm64JITCore;
class X86JITCore;
class InterpreterCore;
class Dispatcher;
}
namespace HLE {
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
}
}
namespace FEXCore::IR {
class RegisterAllocationPass;
class RegisterAllocationData;
class IRListView;
namespace Validation {
@@ -79,7 +61,6 @@ namespace FEXCore::Context {
friend class FEXCore::CPU::X86JITCore;
#endif
friend class FEXCore::CPU::InterpreterCore;
friend class FEXCore::IR::Validation::IRValidation;
struct {
@@ -94,39 +75,28 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(ABINoPF, ABINOPF);
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(Core, CORE);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
FEX_CONFIG_OPT(DumpIR, DUMPIR);
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(x86dec_SynchronizeRIPOnAllBlocks, X86DEC_SYNCHRONIZERIPONALLBLOCKS);
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
} Config;
using IntCallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
IntCallbackReturn InterpreterCallbackReturn;
FEXCore::HostFeatures HostFeatures;
std::mutex ThreadCreationMutex;
FEXCore::Core::InternalThreadState* ParentThread{};
uint64_t ThreadID{};
FEXCore::Core::InternalThreadState* ParentThread;
std::vector<FEXCore::Core::InternalThreadState*> Threads;
std::atomic_bool CoreShuttingDown{false};
bool NeedToCheckXID{true};
std::mutex IdleWaitMutex;
std::condition_variable IdleWaitCV;
@@ -135,16 +105,34 @@ namespace FEXCore::Context {
Event PauseWait;
bool Running{};
std::shared_mutex CodeInvalidationMutex;
FEXCore::CPUIDEmu CPUID;
FEXCore::HLE::SyscallHandler *SyscallHandler{};
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CustomCPUFactoryType CustomCPUFactory;
FEXCore::Context::ExitHandler CustomExitHandler;
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> CustomExitHandler;
struct AOTIRCacheEntry {
uint64_t start;
uint64_t len;
uint64_t crc;
IR::IRListView *IR;
IR::RegisterAllocationData *RAData;
};
std::function<std::unique_ptr<std::istream>(const std::string&)> AOTIRLoader;
std::unordered_map<std::string, std::map<uint64_t, AOTIRCacheEntry>> AOTIRCache;
struct AddrToFileEntry {
uint64_t Start;
uint64_t Len;
uint64_t Offset;
std::string fileid;
void *CachedFileEntry;
};
std::map<uint64_t, AddrToFileEntry> AddrToFile;
#ifdef BLOCKSTATS
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
@@ -156,9 +144,9 @@ namespace FEXCore::Context {
Context();
~Context();
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer);
bool InitCore(FEXCore::CodeLoader *Loader);
FEXCore::Context::ExitReason RunUntilExit();
int GetProgramStatus() const;
int GetProgramStatus();
bool IsPaused() const { return !Running; }
void Pause();
void Run();
@@ -169,40 +157,20 @@ namespace FEXCore::Context {
void StopThread(FEXCore::Core::InternalThreadState *Thread);
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
bool GetGdbServerStatus() { return (bool)DebugServer; }
void StartGdbServer();
void StopGdbServer();
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
void HandleCallback(uint64_t RIP);
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
static void RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
FHU::ScopedSignalMaskWithSharedLock lk(Frame->Thread->CTX->CodeInvalidationMutex);
return Fn(Frame, record);
// Wrapper which takes CpuStateFrame instead of InternalThreadState
static void RemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
RemoveCodeEntry(Frame->Thread, GuestRIP);
}
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
// Must be called from owning thread
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
// returns false if a handler was already registered
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data);
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
// Debugger interface
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
uint64_t GetThreadCount() const;
@@ -210,172 +178,65 @@ namespace FEXCore::Context {
bool GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data);
bool FindHostCodeForRIP(uint64_t RIP, uint8_t **Code);
struct GenerateIRResult {
FEXCore::IR::IRListView* IRList;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
uint64_t TotalInstructions;
uint64_t TotalInstructionsLength;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
// XXX:
// bool FindIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir);
// void SetIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir);
void LoadEntryList();
struct CompileCodeResult {
void* CompiledCode;
FEXCore::IR::IRListView* IRData;
FEXCore::Core::DebugData* DebugData;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
bool GeneratedIR;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
std::tuple<FEXCore::IR::IRListView *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t, uint64_t, uint64_t> GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
std::tuple<void *, FEXCore::IR::IRListView *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool, uint64_t, uint64_t> CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
// same as CompileBlock, but aborts on failure
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
bool LoadAOTIRCache(std::istream &stream);
bool WriteAOTIRCache(std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
// Used for thread creation from syscalls
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param NewThreadState The initial thread state to setup for our state
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* OS thread Creation:
* - Thread = CreateThread(NewState, PPID);
* - InitializeThread(Thread);
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
/**
* @brief Initializes TID, PID and TLS data for a thread
*
* @param Thread The internal FEX thread state object
*/
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
*
* @param Thread The internal FEX thread state object
*
* The OS thread will wait until RunThread is executed
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
void InitializeThread(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Starts the OS thread object to start executing guest code
*
* @param Thread The internal FEX thread state object
*/
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
void RunThread(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Destroys this FEX thread object and stops tracking it internally
*
* @param Thread The internal FEX thread state object
*/
void DestroyThread(FEXCore::Core::InternalThreadState *Thread);
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
void CleanupAfterFork(FEXCore::Core::InternalThreadState *ExceptForThread);
std::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
std::vector<FEXCore::Core::InternalThreadState*> *const GetThreads() { return &Threads; }
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const std::string &filename);
void UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry);
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
#if ENABLE_JITSYMBOLS
FEXCore::JITSymbols Symbols;
#endif
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread);
void FinalizeAOTIRCache() {
IRCaptureCache.FinalizeAOTIRCache();
}
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void SetAOTIRLoader(std::function<int(const std::string&)> CacheReader) {
IRCaptureCache.SetAOTIRLoader(CacheReader);
}
void SetAOTIRWriter(std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
IRCaptureCache.SetAOTIRWriter(CacheWriter);
}
void SetAOTIRRenamer(std::function<void(const std::string&)> CacheRenamer) {
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
}
void AppendThunkDefinitions(std::vector<FEXCore::IR::ThunkDefinition> const& Definitions);
FEXCore::Utils::PooledAllocatorMMap OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorMMap FrontendAllocator;
void MarkMemoryShared();
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
protected:
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
private:
/**
* @brief Does some final thread initialization
*
* @param Thread The internal FEX thread state object
*
* InitCore and CreateThread both call this to finish up thread object initialization
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Initializes the JIT compilers for the thread
*
* @param State The internal FEX thread state object
*
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
void WaitForIdleWithTimeout();
void NotifyPause();
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length);
FEXCore::CodeLoader *LocalLoader{};
// Entry Cache
std::optional<std::string> GetFilenameHash(std::string const &Filename) const;
void AddThreadRIPsToEntryList(FEXCore::Core::InternalThreadState *Thread);
void SaveEntryList();
std::set<uint64_t> EntryList;
std::vector<uint64_t> InitLocations;
uint64_t StartingRIP;
std::mutex ExitMutex;
std::unique_ptr<GdbServer> DebugServer;
IR::AOTIRCaptureCache IRCaptureCache;
std::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
bool StartPaused = false;
bool IsMemoryShared = false;
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
std::shared_mutex CustomIRMutex;
std::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
FEXCore::CPU::DispatcherConfig DispatcherConfig;
};
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
File diff suppressed because it is too large. Load diff
@@ -12,58 +12,6 @@ namespace FEXCore::ArchHelpers::Arm64 {
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
constexpr uint32_t RCPC2_MASK = 0x3F'E0'0C'00;
constexpr uint32_t LDAPUR_INST = 0x19'40'00'00;
constexpr uint32_t STLUR_INST = 0x19'00'00'00;
constexpr uint32_t LDAXP_MASK = 0xBF'FF'80'00;
constexpr uint32_t LDAXP_INST = 0x88'7F'80'00;
constexpr uint32_t STLXP_MASK = 0xBF'E0'80'00;
constexpr uint32_t STLXP_INST = 0x88'20'80'00;
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
constexpr uint32_t CBNZ_MASK = 0x7F'00'00'00;
constexpr uint32_t CBNZ_INST = 0x35'00'00'00;
constexpr uint32_t ALU_OP_MASK = 0x7F'20'00'00;
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
constexpr uint32_t ADD_SHIFT_INST = 0x0B'20'00'00;
constexpr uint32_t SUB_SHIFT_INST = 0x4B'20'00'00;
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
constexpr uint32_t CMP_SHIFT_INST = 0x6B'20'00'00;
constexpr uint32_t AND_INST = 0x0A'00'00'00;
constexpr uint32_t BIC_INST = 0x0A'20'00'00;
constexpr uint32_t OR_INST = 0x2A'00'00'00;
constexpr uint32_t ORN_INST = 0x2A'20'00'00;
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
constexpr uint32_t EON_INST = 0x4A'20'00'00;
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
constexpr uint32_t CLREX_MASK = 0xFF'FF'F0'FF;
constexpr uint32_t CLREX_INST = 0xD5'03'30'5F;
enum ExclusiveAtomicPairType {
TYPE_SWAP,
TYPE_ADD,
TYPE_SUB,
TYPE_AND,
TYPE_BIC,
TYPE_OR,
TYPE_ORN,
TYPE_EOR,
TYPE_EON,
TYPE_NEG, // This is just a sub with zero. Need to know the differences
};
// Load ops are 4 bits
// Acquire and release bits are independent on the instruction
constexpr uint32_t ATOMIC_ADD_OP = 0b0000;
@@ -76,34 +24,7 @@ namespace FEXCore::ArchHelpers::Arm64 {
constexpr uint32_t ATOMIC_UMIN_OP = 0b0111;
constexpr uint32_t ATOMIC_SWAP_OP = 0b1000;
constexpr uint32_t REGISTER_MASK = 0b11111;
constexpr uint32_t RD_OFFSET = 0;
constexpr uint32_t RN_OFFSET = 5;
constexpr uint32_t RM_OFFSET = 16;
constexpr uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
0b1011'0000'0000; // Inner shareable all
inline uint32_t GetRdReg(uint32_t Instr) {
return (Instr >> RD_OFFSET) & REGISTER_MASK;
}
inline uint32_t GetRnReg(uint32_t Instr) {
return (Instr >> RN_OFFSET) & REGISTER_MASK;
}
inline uint32_t GetRmReg(uint32_t Instr) {
return (Instr >> RM_OFFSET) & REGISTER_MASK;
}
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr);
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info);
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr);
uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr);
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr);
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr);
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr);
[[nodiscard]] bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext);
}
@@ -1,109 +1,50 @@
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Context/Context.h"
#include "Interface/HLE/Thunks/Thunks.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Core/CoreState.h>
#include <aarch64/cpu-aarch64.h>
#include <aarch64/instructions-aarch64.h>
#include <cpu-features.h>
#include <utils-vixl.h>
#include <array>
#include <tuple>
#include <utility>
#include "aarch64/cpu-aarch64.h"
namespace FEXCore::CPU {
#define STATE x28
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
: Emitter(size ? (uint8_t*)FEXCore::Allocator::mmap(nullptr, size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0) : nullptr, size)
, EmitterCTX {ctx} {
Arm64Emitter::Arm64Emitter(size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
CPU.SetUp();
}
Arm64Emitter::~Arm64Emitter() {
auto BufferSize = GetBufferSize();
if (BufferSize) {
FEXCore::Allocator::munmap(GetBufferBase(), BufferSize);
auto Features = vixl::CPUFeatures::InferFromOS();
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
// RCPC is bugged on Snapdragon 865
// Causes glibc cond16 test to immediately throw assert
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
if (SupportsAtomics) {
// Hypervisor can hide this on the c630?
Features.Combine(vixl::CPUFeatures::Feature::kLORegions);
}
SetCPUFeatures(Features);
if (!SupportsAtomics) {
WARN_ONCE("Host CPU doesn't support atomics. Expect bad performance");
}
}
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) {
bool Is64Bit = Reg.IsX();
int Segments = Is64Bit ? 4 : 2;
if (Is64Bit && ((~Constant)>> 16) == 0) {
movn(s, Reg, (~Constant) & 0xFFFF);
if (NOPPad) {
nop(); nop(); nop();
}
movn(Reg, (~Constant) & 0xFFFF);
return;
}
int NumMoves = 1;
int RequiredMoveSegments{};
// Count the number of move segments
// We only want to use ADRP+ADD if we have more than 1 segment
for (size_t i = 0; i < Segments; ++i) {
movz(Reg, (Constant) & 0xFFFF, 0);
for (int i = 1; i < Segments; ++i) {
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
if (Part != 0) {
++RequiredMoveSegments;
}
}
// ADRP+ADD is specifically optimized in hardware
// Check if we can use this
auto PC = GetCursorAddress<uint64_t>();
// PC aligned to page
uint64_t AlignedPC = PC & ~0xFFFULL;
// Offset from aligned PC
int64_t AlignedOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(AlignedPC);
// If the aligned offset is within the 4GB window then we can use ADRP+ADD
// and the number of move segments more than 1
if (RequiredMoveSegments > 1 && vixl::IsInt32(AlignedOffset)) {
// If this is 4k page aligned then we only need ADRP
if ((AlignedOffset & 0xFFF) == 0) {
adrp(Reg, AlignedOffset >> 12);
}
else {
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
// 21-bit signed integer here
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
if (vixl::IsInt21(SmallOffset)) {
adr(Reg, SmallOffset);
}
else {
// Need to use ADRP + ADD
adrp(Reg, AlignedOffset >> 12);
add(s, Reg, Reg, Constant & 0xFFF);
NumMoves = 2;
}
}
}
else {
movz(s, Reg, (Constant) & 0xFFFF, 0);
for (int i = 1; i < Segments; ++i) {
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
if (Part) {
movk(s, Reg, Part, i * 16);
++NumMoves;
}
}
}
if (NOPPad) {
for (int i = NumMoves; i < Segments; ++i) {
nop();
if (Part) {
movk(Reg, Part, i * 16);
}
}
}
@@ -111,288 +52,168 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
void Arm64Emitter::PushCalleeSavedRegisters() {
// We need to save pairs of registers
// We save r19-r30
const std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
{ARMEmitter::XReg::x21, ARMEmitter::XReg::x22},
{ARMEmitter::XReg::x23, ARMEmitter::XReg::x24},
{ARMEmitter::XReg::x25, ARMEmitter::XReg::x26},
{ARMEmitter::XReg::x27, ARMEmitter::XReg::x28},
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
MemOperand PairOffset(sp, -16, PreIndex);
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
{x19, x20},
{x21, x22},
{x23, x24},
{x25, x26},
{x27, x28},
{x29, x30},
}};
for (auto &RegPair : CalleeSaved) {
stp<ARMEmitter::IndexType::PRE>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, -16);
stp(RegPair.first, RegPair.second, PairOffset);
}
// Additionally we need to store the lower 64bits of v8-v15
// Here's a fun thing, we can use two ST4 instructions to store everything
// We just need a single sub to sp before that
const std::array<
std::tuple<ARMEmitter::DRegister,
ARMEmitter::DRegister,
ARMEmitter::DRegister,
ARMEmitter::DRegister>, 2> FPRs = {{
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
std::tuple<vixl::aarch64::VRegister,
vixl::aarch64::VRegister,
vixl::aarch64::VRegister,
vixl::aarch64::VRegister>, 2> FPRs = {{
{v8, v9, v10, v11},
{v12, v13, v14, v15},
}};
uint32_t VectorSaveSize = sizeof(uint64_t) * 8;
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, VectorSaveSize);
sub(sp, sp, VectorSaveSize);
// SP supporting move
// We just saved x19 so it is safe
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r19, ARMEmitter::Reg::rsp, 0);
add(x19, sp, 0);
MemOperand QuadOffset(x19, 32, PostIndex);
for (auto &RegQuad : FPRs) {
st4(ARMEmitter::SubRegSize::i64Bit,
std::get<0>(RegQuad),
std::get<1>(RegQuad),
std::get<2>(RegQuad),
std::get<3>(RegQuad),
st4(std::get<0>(RegQuad).D(),
std::get<1>(RegQuad).D(),
std::get<2>(RegQuad).D(),
std::get<3>(RegQuad).D(),
0,
ARMEmitter::Reg::r19,
32);
QuadOffset);
}
}
void Arm64Emitter::PopCalleeSavedRegisters() {
const std::array<
std::tuple<ARMEmitter::DRegister,
ARMEmitter::DRegister,
ARMEmitter::DRegister,
ARMEmitter::DRegister>, 2> FPRs = {{
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
std::tuple<vixl::aarch64::VRegister,
vixl::aarch64::VRegister,
vixl::aarch64::VRegister,
vixl::aarch64::VRegister>, 2> FPRs = {{
{v12, v13, v14, v15},
{v8, v9, v10, v11},
}};
MemOperand QuadOffset(sp, 32, PostIndex);
for (auto &RegQuad : FPRs) {
ld4(ARMEmitter::SubRegSize::i64Bit,
std::get<0>(RegQuad),
std::get<1>(RegQuad),
std::get<2>(RegQuad),
std::get<3>(RegQuad),
ld4(std::get<0>(RegQuad).D(),
std::get<1>(RegQuad).D(),
std::get<2>(RegQuad).D(),
std::get<3>(RegQuad).D(),
0,
ARMEmitter::Reg::rsp,
32);
QuadOffset);
}
const std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
{ARMEmitter::XReg::x27, ARMEmitter::XReg::x28},
{ARMEmitter::XReg::x25, ARMEmitter::XReg::x26},
{ARMEmitter::XReg::x23, ARMEmitter::XReg::x24},
{ARMEmitter::XReg::x21, ARMEmitter::XReg::x22},
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
MemOperand PairOffset(sp, 16, PostIndex);
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
{x29, x30},
{x27, x28},
{x25, x26},
{x23, x24},
{x21, x22},
{x19, x20},
}};
for (auto &RegPair : CalleeSaved) {
ldp<ARMEmitter::IndexType::POST>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, 16);
ldp(RegPair.first, RegPair.second, PairOffset);
}
}
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
if (!StaticRegisterAllocation()) {
return;
}
void Arm64Emitter::SpillStaticRegs() {
for (size_t i = 0; i < SRA64.size(); i+=2) {
auto Reg1 = SRA64[i];
auto Reg2 = SRA64[i+1];
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
((1U << Reg2.Idx()) & GPRSpillMask)) {
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
}
else if (((1U << Reg1.Idx()) & GPRSpillMask)) {
str(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
}
else if (((1U << Reg2.Idx()) & GPRSpillMask)) {
str(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
}
stp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
if (FPRs) {
if (EmitterCTX->HostFeatures.SupportsAVX) {
for (size_t i = 0; i < SRAFPR.size(); i++) {
const auto Reg = SRAFPR[i];
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
st1b<ARMEmitter::SubRegSize::i8Bit>(Reg, PRED_TMP_32B, STATE.R(), TMP4.R());
}
}
} else {
if (GPRSpillMask && FPRSpillMask == ~0U) {
// Optimize the common case where we can spill four registers per instruction
auto TmpReg = SRA64[__builtin_ffs(GPRSpillMask)];
// Load the sse offset in to the temporary register
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
for (size_t i = 0; i < SRAFPR.size(); i += 4) {
const auto Reg1 = SRAFPR[i];
const auto Reg2 = SRAFPR[i + 1];
const auto Reg3 = SRAFPR[i + 2];
const auto Reg4 = SRAFPR[i + 3];
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
}
}
else {
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
const auto Reg1 = SRAFPR[i];
const auto Reg2 = SRAFPR[i + 1];
if (((1U << Reg1.Idx()) & FPRSpillMask) &&
((1U << Reg2.Idx()) & FPRSpillMask)) {
stp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
}
else if (((1U << Reg1.Idx()) & FPRSpillMask)) {
str(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
}
else if (((1U << Reg2.Idx()) & FPRSpillMask)) {
str(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0]));
}
}
}
}
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
stp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
}
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
if (!StaticRegisterAllocation()) {
return;
}
if (FPRs) {
if (EmitterCTX->HostFeatures.SupportsAVX) {
// Set up predicate registers.
// We don't bother spilling these in SpillStaticRegs,
// since all that matters is we restore them on a fill.
// It's not a concern if they get trounced by something else.
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
for (size_t i = 0; i < SRAFPR.size(); i++) {
const auto Reg = SRAFPR[i];
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg, PRED_TMP_32B, STATE.R(), TMP4.R());
}
}
} else {
if (GPRFillMask && FPRFillMask == ~0U) {
// Optimize the common case where we can fill four registers per instruction.
// Use one of the filling static registers before we fill it.
auto TmpReg = SRA64[__builtin_ffs(GPRFillMask)];
// Load the sse offset in to the temporary register
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
for (size_t i = 0; i < SRAFPR.size(); i += 4) {
const auto Reg1 = SRAFPR[i];
const auto Reg2 = SRAFPR[i + 1];
const auto Reg3 = SRAFPR[i + 2];
const auto Reg4 = SRAFPR[i + 3];
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
}
}
else {
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
const auto Reg1 = SRAFPR[i];
const auto Reg2 = SRAFPR[i + 1];
if (((1U << Reg1.Idx()) & FPRFillMask) &&
((1U << Reg2.Idx()) & FPRFillMask)) {
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
}
else if (((1U << Reg1.Idx()) & FPRFillMask)) {
ldr(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
}
else if (((1U << Reg2.Idx()) & FPRFillMask)) {
ldr(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0]));
}
}
}
}
}
void Arm64Emitter::FillStaticRegs() {
for (size_t i = 0; i < SRA64.size(); i+=2) {
auto Reg1 = SRA64[i];
auto Reg2 = SRA64[i+1];
if (((1U << Reg1.Idx()) & GPRFillMask) &&
((1U << Reg2.Idx()) & GPRFillMask)) {
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
}
else if ((1U << Reg1.Idx()) & GPRFillMask) {
ldr(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
}
else if ((1U << Reg2.Idx()) & GPRFillMask) {
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
}
ldp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
ldp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
}
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
const auto GPRSize = 1 * Core::CPUState::GPR_REG_SIZE;
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
: Core::CPUState::XMM_SSE_REG_SIZE;
const auto FPRSize = RAFPR.size() * FPRRegSize;
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
void Arm64Emitter::PushDynamicRegsAndLR() {
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
sub(sp, sp, SPOffset);
int i = 0;
// rsp capable move
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
if (CanUseSVE) {
for (size_t i = 0; i < RAFPR.size(); i += 4) {
const auto Reg1 = RAFPR[i];
const auto Reg2 = RAFPR[i + 1];
const auto Reg3 = RAFPR[i + 2];
const auto Reg4 = RAFPR[i + 3];
st4b(Reg1, Reg2, Reg3, Reg4, PRED_TMP_32B, TmpReg, 0);
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
}
} else {
static_assert(RAFPR.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
for (size_t i = 0; i < RAFPR.size(); i += 4) {
const auto Reg1 = RAFPR[i];
const auto Reg2 = RAFPR[i + 1];
const auto Reg3 = RAFPR[i + 2];
const auto Reg4 = RAFPR[i + 3];
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
}
for (auto RA : RAFPR)
{
str(RA.Q(), MemOperand(sp, i * 8));
i+=2;
}
str(ARMEmitter::XReg::lr, TmpReg, 0);
#if 0 // All GPRs should be caller saved
for (auto RA : RA64)
{
str(RA, MemOperand(sp, i * 8));
i++;
}
#endif
str(lr, MemOperand(sp, i * 8));
}
void Arm64Emitter::PopDynamicRegsAndLR() {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
int i = 0;
if (CanUseSVE) {
for (size_t i = 0; i < RAFPR.size(); i += 4) {
const auto Reg1 = RAFPR[i];
const auto Reg2 = RAFPR[i + 1];
const auto Reg3 = RAFPR[i + 2];
const auto Reg4 = RAFPR[i + 3];
ld4b(Reg1, Reg2, Reg3, Reg4, PRED_TMP_32B, ARMEmitter::Reg::rsp);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
}
} else {
for (size_t i = 0; i < RAFPR.size(); i += 4) {
const auto Reg1 = RAFPR[i];
const auto Reg2 = RAFPR[i + 1];
const auto Reg3 = RAFPR[i + 2];
const auto Reg4 = RAFPR[i + 3];
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
}
for (auto RA : RAFPR)
{
ldr(RA.Q(), MemOperand(sp, i * 8));
i+=2;
}
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
#if 0 // All GPRs should be caller saved
for (auto RA : RA64)
{
ldr(RA, MemOperand(sp, i * 8));
i++;
}
#endif
ldr(lr, MemOperand(sp, i * 8));
add(sp, sp, SPOffset);
}
void Arm64Emitter::ResetStack() {
if (SpillSlots == 0)
return;
if (IsImmAddSub(SpillSlots * 16)) {
add(sp, sp, SpillSlots * 16);
} else {
// Too big to fit in a 12bit immediate
LoadConstant(x0, SpillSlots * 16);
add(sp, sp, x0);
}
}
void Arm64Emitter::Align16B() {
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
uint64_t CurrentOffset = GetBuffer()->GetOffsetAddress<uint64_t>(GetCursorOffset());
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
nop();
nop();
}
}
@@ -1,188 +1,74 @@
#pragma once
#include "FEXCore/Utils/EnumUtils.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/ObjectCache/Relocations.h"
#include <aarch64/assembler-aarch64.h>
#include <aarch64/constants-aarch64.h>
#include <aarch64/cpu-aarch64.h>
#include <aarch64/operands-aarch64.h>
#include <platform-vixl.h>
#ifdef VIXL_DISASSEMBLER
#include <aarch64/disasm-aarch64.h>
#endif
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#include <aarch64/simulator-constants-aarch64.h>
#endif
#include <FEXCore/Config/Config.h>
#include <array>
#include <cstddef>
#include <cstdint>
#include <utility>
#include "aarch64/assembler-aarch64.h"
#include "aarch64/cpu-aarch64.h"
namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
// All but x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA64 = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9, FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r18, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r15, FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r13, FEXCore::ARMEmitter::Reg::r29
const std::array<aarch64::Register, 16> SRA64 = {
x4, x5, x6, x7, x8, x9, x10, x11,
x12, x18, x17, x16, x15, x14, x13, x29
};
// All are callee saved
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA64 = {
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23, FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
FEXCore::ARMEmitter::Reg::r19
const std::array<aarch64::Register, 9> RA64 = {
x20, x21, x22, x23, x24, x25, x26, x27,
x19
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RA64Pair = {{
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
const std::array<std::pair<aarch64::Register, aarch64::Register>, 4> RA64Pair = {{
{x20, x21},
{x22, x23},
{x24, x25},
{x26, x27},
}};
const std::array<std::pair<aarch64::Register, aarch64::Register>, 4> RA32Pair = {{
{w20, w21},
{w22, w23},
{w24, w25},
{w26, w27},
}};
// All are caller saved
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19, FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27, FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
const std::array<aarch64::VRegister, 16> SRAFPR = {
v16, v17, v18, v19, v20, v21, v22, v23,
v24, v25, v26, v27, v28, v29, v30, v31
};
// v8..v15 = (lower 64bits) Callee saved
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
/*FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1, FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,*/FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, // FEXCore::ARMEmitter::VReg::v0 ~ FEXCore::ARMEmitter::VReg::v3 are used as temps
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9, FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13, FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15
const std::array<aarch64::VRegister, 12> RAFPR = {
/*v0, v1, v2, v3,*/v4, v5, v6, v7, // v0 ~ v3 are used as temps
v8, v9, v10, v11, v12, v13, v14, v15
};
// Contains the address to the currently available CPU state
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
// GPR temporaries. Only x3 can be used across spill boundaries
// so if these ever need to change, be very careful about that.
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x0;
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x1;
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x2;
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x3;
// Vector temporaries
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v0;
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
constexpr auto VTMP3 = FEXCore::ARMEmitter::VReg::v2;
constexpr auto VTMP4 = FEXCore::ARMEmitter::VReg::v3;
// Predicate register temporaries (used when AVX support is enabled)
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
class Arm64Emitter : public vixl::aarch64::Assembler {
protected:
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
~Arm64Emitter();
Arm64Emitter(size_t size);
FEXCore::Context::Context *EmitterCTX;
vixl::aarch64::CPU CPU;
void LoadConstant(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
bool SupportsAtomics{};
bool SupportsRCPC{};
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant);
void SpillStaticRegs();
void FillStaticRegs();
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
// TMP4 is left alone.
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
static constexpr uint32_t CALLER_GPR_MASK = 0b0011'1111'1111'1111'1111;
// This isn't technically true because the lower 64-bits of v8..v15 are callee saved
// We can't guarantee only the lower 64bits are used so flush everything
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
void PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg);
void PushDynamicRegsAndLR();
void PopDynamicRegsAndLR();
void PushCalleeSavedRegisters();
void PopCalleeSavedRegisters();
void ResetStack();
void Align16B();
#ifdef VIXL_SIMULATOR
// Generates a vixl simulator runtime call.
//
// This matches behaviour of vixl's macro assembler, but we need to reimplement it since we aren't using the macro assembler.
// This isn't too complex with how vixl emits this.
//
// Emit:
// 1) hlt(kRuntimeCallOpcode)
// 2) Simulator wrapper handler
// 3) Function to call
// 4) Style of the function call (Call versus tail-call)
template<typename R, typename... P>
void GenerateRuntimeCall(R (*Function)(P...)) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
uintptr_t FunctionAddress = reinterpret_cast<uintptr_t>(Function);
hlt(vixl::aarch64::kRuntimeCallOpcode);
// Simulator wrapper address pointer.
dc64(SimulatorWrapperAddress);
// Runtime function address to call
dc64(FunctionAddress);
// Call type
dc32(vixl::aarch64::kCallRuntime);
}
template<typename R, typename... P>
void GenerateIndirectRuntimeCall(ARMEmitter::Register Reg) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
// Simulator wrapper address pointer.
dc64(SimulatorWrapperAddress);
// Register that contains the function to call
dc32(Reg.Idx());
// Call type
dc32(vixl::aarch64::kCallRuntime);
}
template<>
void GenerateIndirectRuntimeCall<float, __uint128_t>(ARMEmitter::Register Reg) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
// Simulator wrapper address pointer.
dc64(SimulatorWrapperAddress);
// Register that contains the function to call
dc32(Reg.Idx());
// Call type
dc32(vixl::aarch64::kCallRuntime);
}
#endif
#ifdef VIXL_DISASSEMBLER
vixl::aarch64::PrintDisassembler Disasm {stderr};
#endif
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
uint32_t SpillSlots{};
};
}
@@ -1,7 +1,6 @@
#include "Interface/Core/ArchHelpers/Arm64.h"
#include <FEXCore/Utils/LogManager.h>
#include <stdint.h>
namespace FEXCore::ArchHelpers::Arm64 {
@@ -12,16 +11,16 @@ namespace FEXCore::ArchHelpers::Arm64 {
// Obvously such a configuration can't do the actual arm64-specific stuff
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
ERROR_AND_DIE_FMT("HandleCASPAL Not Implemented");
ERROR_AND_DIE("HandleCASPAL Not Implemented");
}
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
ERROR_AND_DIE_FMT("HandleCASAL Not Implemented");
ERROR_AND_DIE("HandleCASAL Not Implemented");
}
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
ERROR_AND_DIE("HandleAtomicMemOp Not Implemented");
}
#endif
}
}
@@ -1,992 +0,0 @@
/* ALU instruction emitters.
*
* Almost all of these operations have `ARMEmitter::Size` as their first argument.
* This allows both 32-bit and 64-bit selection of how that instruction is going to operate.
*
* Some emitter operations explicitly use `XRegister` or `WRegister`.
* This is usually due to the instruction only supporting one operating size.
* Although in some cases is a minor convenience without any performance implications.
*
* FEX-Emu ALU operations usually have a 32-bit or 64-bit operating size encoded in the IR operation,
* This allows FEX to use a single helper function which decodes to both handlers.
*/
private:
static bool IsADRRange(int64_t Imm) {
return Imm >= -1048576 && Imm <= 1048575;
}
static bool IsADRPRange(int64_t Imm) {
return Imm >= -4294967296 && Imm <= 4294963200;
}
static bool IsADRPAligned(int64_t Imm) {
return (Imm & 0xFFF) == 0;
}
public:
// PC relative
void adr(FEXCore::ARMEmitter::Register rd, uint32_t Imm) {
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
void adr(FEXCore::ARMEmitter::Register rd, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
void adr(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADR });
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
}
void adr(FEXCore::ARMEmitter::Register rd, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
adr(rd, &Label->Backward);
}
else {
adr(rd, &Label->Forward);
}
}
void adrp(FEXCore::ARMEmitter::Register rd, uint32_t Imm) {
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
void adrp(FEXCore::ARMEmitter::Register rd, BackwardLabel const* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
void adrp(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADRP });
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
}
void adrp(FEXCore::ARMEmitter::Register rd, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
adrp(rd, &Label->Backward);
}
else {
adrp(rd, &Label->Forward);
}
}
void LongAddressGen(FEXCore::ARMEmitter::Register rd, BackwardLabel const* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
if (IsADRRange(Imm)) {
// If the range is in ADR range then we can just use ADR.
adr(rd, Label);
}
else if (IsADRPRange(Imm)) {
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL)
- (GetCursorAddress<int64_t>() & ~0xFFFLL);
// If the range is in the ADRP range then we can use ADRP.
bool NeedsOffset = !IsADRPAligned(reinterpret_cast<uint64_t>(Label->Location));
uint64_t AlignedOffset = reinterpret_cast<uint64_t>(Label->Location) & 0xFFFULL;
// First emit ADRP
adrp(rd, ADRPImm >> 12);
if (NeedsOffset) {
// Now even an add
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
}
}
else {
LOGMAN_MSG_A_FMT("Unscaled offset too large");
FEX_UNREACHABLE;
}
}
void LongAddressGen(FEXCore::ARMEmitter::Register rd, ForwardLabel* Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN });
// Emit a register index and a nop. These will be backpatched.
dc32(rd.Idx());
nop();
}
void LongAddressGen(FEXCore::ARMEmitter::Register rd, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
LongAddressGen(rd, &Label->Backward);
}
else {
LongAddressGen(rd, &Label->Forward);
}
}
// Add/subtract immediate
void add(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
constexpr uint32_t Op = 0b0001'0001'0 << 23;
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
}
void adds(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
constexpr uint32_t Op = 0b0011'0001'0 << 23;
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
}
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
constexpr uint32_t Op = 0b0101'0001'0 << 23;
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
}
void cmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
constexpr uint32_t Op = 0b0111'0001'0 << 23;
DataProcessing_AddSub_Imm(Op, s, FEXCore::ARMEmitter::Reg::rsp, rn, Imm, LSL12);
}
void subs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
constexpr uint32_t Op = 0b0111'0001'0 << 23;
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
}
// Logical immediate
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
[[maybe_unused]] const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Imm,
RegSizeInBits(s),
&n,
&imms,
&immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
and_(s, rd, rn, n, immr, imms);
}
void bic(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
and_(s, rd, rn, ~Imm);
}
void ands(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
[[maybe_unused]] const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Imm,
RegSizeInBits(s),
&n,
&imms,
&immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
ands(s, rd, rn, n, immr, imms);
}
void bics(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
ands(s, rd, rn, ~Imm);
}
void orr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
[[maybe_unused]] const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Imm,
RegSizeInBits(s),
&n,
&imms,
&immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
orr(s, rd, rn, n, immr, imms);
}
void eor(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
[[maybe_unused]] const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Imm,
RegSizeInBits(s),
&n,
&imms,
&immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
eor(s, rd, rn, n, immr, imms);
}
// Move wide immediate
void movn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset = 0) {
LOGMAN_THROW_A_FMT((Imm & 0xFFFF0000U) == 0, "Upper bits of move wide not valid");
LOGMAN_THROW_A_FMT((Offset % 16) == 0, "Offset must be 16bit aligned");
constexpr uint32_t Op = 0b001'0010'100 << 21;
DataProcessing_MoveWide(Op, s, rd, Imm, Offset >> 4);
}
void mov(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm) {
movz(s, rd, Imm, 0);
}
void mov(FEXCore::ARMEmitter::XRegister rd, uint32_t Imm) {
movz(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), Imm, 0);
}
void mov(FEXCore::ARMEmitter::WRegister rd, uint32_t Imm) {
movz(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), Imm, 0);
}
void movz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset = 0) {
LOGMAN_THROW_A_FMT((Imm & 0xFFFF0000U) == 0, "Upper bits of move wide not valid");
LOGMAN_THROW_A_FMT((Offset % 16) == 0, "Offset must be 16bit aligned");
constexpr uint32_t Op = 0b101'0010'100 << 21;
DataProcessing_MoveWide(Op, s, rd, Imm, Offset >> 4);
}
void movk(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset = 0) {
LOGMAN_THROW_A_FMT((Imm & 0xFFFF0000U) == 0, "Upper bits of move wide not valid");
LOGMAN_THROW_A_FMT((Offset % 16) == 0, "Offset must be 16bit aligned");
constexpr uint32_t Op = 0b111'0010'100 << 21;
DataProcessing_MoveWide(Op, s, rd, Imm, Offset >> 4);
}
void movn(FEXCore::ARMEmitter::XRegister rd, uint32_t Imm, uint32_t Offset = 0) {
movn(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), Imm, Offset);
}
void movz(FEXCore::ARMEmitter::XRegister rd, uint32_t Imm, uint32_t Offset = 0) {
movz(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), Imm, Offset);
}
void movk(FEXCore::ARMEmitter::XRegister rd, uint32_t Imm, uint32_t Offset = 0) {
movk(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), Imm, Offset);
}
void movn(FEXCore::ARMEmitter::WRegister rd, uint32_t Imm, uint32_t Offset = 0) {
movn(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), Imm, Offset);
}
void movz(FEXCore::ARMEmitter::WRegister rd, uint32_t Imm, uint32_t Offset = 0) {
movz(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), Imm, Offset);
}
void movk(FEXCore::ARMEmitter::WRegister rd, uint32_t Imm, uint32_t Offset = 0) {
movk(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), Imm, Offset);
}
// Bitfield
void sxtb(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
sbfm(s, rd, rn, 0, 7);
}
void sxth(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
sbfm(s, rd, rn, 0, 15);
}
void sxtw(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn) {
sbfm(ARMEmitter::Size::i64Bit, rd, rn, 0, 31);
}
void sbfx(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
LOGMAN_THROW_A_FMT(width > 0, "sbfx needs width > 0");
LOGMAN_THROW_A_FMT((lsb + width) <= RegSizeInBits(s), "Tried to sbfx a region larger than the register");
sbfm(s, rd, rn, lsb, lsb + width - 1);
}
void asr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
LOGMAN_THROW_A_FMT(shift <= RegSizeInBits(s), "Tried to asr a region larger than the register");
sbfm(s, rd, rn, shift, RegSizeInBits(s) - 1);
}
void uxtb(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
ubfm(s, rd, rn, 0, 7);
}
void uxth(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
ubfm(s, rd, rn, 0, 15);
}
void uxtw(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
ubfm(s, rd, rn, 0, 31);
}
void ubfm(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b0101'0011'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, s == ARMEmitter::Size::i64Bit, immr, imms);
}
void lsl(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
const auto RegSize = RegSizeInBits(s);
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to asr a region larger than the register");
ubfm(s, rd, rn, (RegSize - shift) % RegSize, RegSize - shift - 1);
}
void lsr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
const auto RegSize = RegSizeInBits(s);
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to asr a region larger than the register");
ubfm(s, rd, rn, shift, RegSize - 1);
}
void ubfx(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
LOGMAN_THROW_A_FMT(width > 0, "ubfx needs width > 0");
LOGMAN_THROW_A_FMT((lsb + width) <= RegSizeInBits(s), "Tried to ubfx a region larger than the register");
ubfm(s, rd, rn, lsb, lsb + width - 1);
}
void bfi(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
const auto RegSize = RegSizeInBits(s);
LOGMAN_THROW_A_FMT(width > 0, "sbfx needs width > 0");
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to sbfx a region larger than the register");
bfm(s, rd, rn, (RegSize - lsb) & (RegSize - 1), width - 1);
}
// Extract
void extr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
constexpr uint32_t Op = 0b001'0011'100 << 21;
LOGMAN_THROW_A_FMT(Imm < RegSizeInBits(s), "Tried to extr a region larger than the register");
DataProcessing_Extract(Op, s, rd, rn, rm, Imm);
}
void ror(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm) {
LOGMAN_THROW_A_FMT(Imm < RegSizeInBits(s), "Tried to extr a region larger than the register");
extr(s, rd, rn, rn, Imm);
}
// Data processing - 2 source
void udiv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0000'10U << 10);
DataProcessing_2Source(Op, s, rd, rn, rm);
}
void sdiv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0000'11U << 10);
DataProcessing_2Source(Op, s, rd, rn, rm);
}
void lslv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0010'00U << 10);
DataProcessing_2Source(Op, s, rd, rn, rm);
}
void lsrv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0010'01U << 10);
DataProcessing_2Source(Op, s, rd, rn, rm);
}
void asrv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0010'10U << 10);
DataProcessing_2Source(Op, s, rd, rn, rm);
}
void rorv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0010'11U << 10);
DataProcessing_2Source(Op, s, rd, rn, rm);
}
void crc32b(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0100'00U << 10);
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
}
void crc32h(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0100'01U << 10);
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
}
void crc32w(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0100'10U << 10);
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
}
void crc32cb(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0101'00U << 10);
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
}
void crc32ch(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0101'01U << 10);
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
}
void crc32cw(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0101'10U << 10);
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
}
void subp(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0000'00U << 10);
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
}
void irg(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0001'00U << 10);
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
}
void gmi(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0001'01U << 10);
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
}
void pacga(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0011'00U << 10);
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
}
void crc32x(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0100'11U << 10);
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
}
void crc32cx(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
constexpr uint32_t Op = (0b001'1010'110U << 21) |
(0b0101'11U << 10);
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
}
void subps(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
constexpr uint32_t Op = (0b011'1010'110U << 21) |
(0b0000'00U << 10);
DataProcessing_2Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm);
}
// Data processing - 1 source
void rbit(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
constexpr uint32_t Op = (0b101'1010'110U << 21) |
(0b0'0000U << 16) |
(0b0000'00U << 10);
DataProcessing_1Source(Op, s, rd, rn);
}
void rev16(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
constexpr uint32_t Op = (0b101'1010'110U << 21) |
(0b0'0000U << 16) |
(0b0000'01U << 10);
DataProcessing_1Source(Op, s, rd, rn);
}
void rev(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn) {
constexpr uint32_t Op = (0b101'1010'110U << 21) |
(0b0'0000U << 16) |
(0b0000'10U << 10);
DataProcessing_1Source(Op, FEXCore::ARMEmitter::Size::i32Bit, rd, rn);
}
void rev32(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn) {
constexpr uint32_t Op = (0b101'1010'110U << 21) |
(0b0'0000U << 16) |
(0b0000'10U << 10);
DataProcessing_1Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn);
}
void clz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
constexpr uint32_t Op = (0b101'1010'110U << 21) |
(0b0'0000U << 16) |
(0b0001'00U << 10);
DataProcessing_1Source(Op, s, rd, rn);
}
void cls(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
constexpr uint32_t Op = (0b101'1010'110U << 21) |
(0b0'0000U << 16) |
(0b0001'01U << 10);
DataProcessing_1Source(Op, s, rd, rn);
}
void rev(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn) {
constexpr uint32_t Op = (0b101'1010'110U << 21) |
(0b0'0000U << 16) |
(0b0000'11U << 10);
DataProcessing_1Source(Op, FEXCore::ARMEmitter::Size::i64Bit, rd, rn);
}
void rev(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
uint32_t Op = (0b101'1010'110U << 21) |
(0b0'0000U << 16) |
(0b0000'10U << 10) |
(s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
DataProcessing_1Source(Op, s, rd, rn);
}
// TODO: PAUTH
// Logical - shifted register
void mov(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
orr(s, rd, FEXCore::ARMEmitter::Reg::zr, rn, ARMEmitter::ShiftType::LSL, 0);
}
void mov(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn) {
orr(FEXCore::ARMEmitter::Size::i64Bit, rd.R(), FEXCore::ARMEmitter::Reg::zr, rn.R(), ARMEmitter::ShiftType::LSL, 0);
}
void mov(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn) {
orr(FEXCore::ARMEmitter::Size::i32Bit, rd.R(), FEXCore::ARMEmitter::Reg::zr, rn.R(), ARMEmitter::ShiftType::LSL, 0);
}
void mvn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
orn(s, rd, FEXCore::ARMEmitter::Reg::zr, rn, Shift, amt);
}
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
constexpr uint32_t Op = 0b000'1010'000U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void ands(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
constexpr uint32_t Op = 0b110'1010'000U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void bic(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
constexpr uint32_t Op = 0b000'1010'001U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void bics(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
constexpr uint32_t Op = 0b110'1010'001U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void orr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
constexpr uint32_t Op = 0b010'1010'000U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void orn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
constexpr uint32_t Op = 0b010'1010'001U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void eor(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
constexpr uint32_t Op = 0b100'1010'000U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void eon(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
constexpr uint32_t Op = 0b100'1010'001U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
// AddSub - shifted register
void add(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
add(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
}
void adds(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
adds(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
}
void sub(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
sub(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
}
void neg(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
sub(rd, FEXCore::ARMEmitter::XReg::zr, rm, Shift, amt);
}
void cmp(FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
subs(ARMEmitter::Size::i64Bit, FEXCore::ARMEmitter::Reg::rsp, rn.R(), rm.R(), Shift, amt);
}
void subs(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
subs(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
}
void negs(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
subs(rd, FEXCore::ARMEmitter::XReg::zr, rm, Shift, amt);
}
void add(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
add(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
}
void adds(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
adds(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
}
void sub(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
sub(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
}
void neg(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
sub(rd, FEXCore::ARMEmitter::WReg::zr, rm, Shift, amt);
}
void cmp(FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
subs(ARMEmitter::Size::i32Bit, FEXCore::ARMEmitter::Reg::rsp, rn.R(), rm.R(), Shift, amt);
}
void subs(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
subs(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
}
void negs(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
subs(rd, FEXCore::ARMEmitter::WReg::zr, rm, Shift, amt);
}
void add(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
constexpr uint32_t Op = 0b000'1011'000U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void adds(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
constexpr uint32_t Op = 0b010'1011'000U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
constexpr uint32_t Op = 0b100'1011'000U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void neg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
sub(s, rd, FEXCore::ARMEmitter::Reg::zr, rm, Shift, amt);
}
void cmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
subs(s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Shift, amt);
}
void subs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
constexpr uint32_t Op = 0b110'1011'000U << 21;
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
}
void negs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
subs(s, rd, FEXCore::ARMEmitter::Reg::zr, rm, Shift, amt);
}
// AddSub - extended register
void add(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
LOGMAN_THROW_AA_FMT(Shift <= 4, "Shift amount is too large");
constexpr uint32_t Op = 0b000'1011'001U << 21;
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
}
void adds(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
constexpr uint32_t Op = 0b010'1011'001U << 21;
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
}
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
constexpr uint32_t Op = 0b100'1011'001U << 21;
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
}
void subs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
constexpr uint32_t Op = 0b110'1011'001U << 21;
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
}
void cmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
constexpr uint32_t Op = 0b110'1011'001U << 21;
DataProcessing_Extended_Reg(Op, s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Option, Shift);
}
// AddSub - with carry
void adc(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = 0b0001'1010'000U << 21;
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
}
void adcs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = 0b0011'1010'000U << 21;
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
}
void sbc(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = 0b0101'1010'000U << 21;
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
}
void sbcs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
constexpr uint32_t Op = 0b0111'1010'000U << 21;
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
}
// Rotate right into flags
void rmif(XRegister rn, uint32_t shift, uint32_t mask) {
LOGMAN_THROW_AA_FMT(shift <= 63, "Shift must be within 0-63. Shift: {}", shift);
LOGMAN_THROW_AA_FMT(mask <= 15, "Mask must be within 0-15. Mask: {}", mask);
uint32_t Op = 0b1011'1010'0000'0000'0000'0100'0000'0000;
Op |= rn.Idx() << 5;
Op |= shift << 15;
Op |= mask;
dc32(Op);
}
// Evaluate into flags
void setf8(WRegister rn) {
constexpr uint32_t Op = 0b0011'1010'0000'0000'0000'1000'0000'1101;
EvaluateIntoFlags(Op, 0, rn);
}
void setf16(WRegister rn) {
constexpr uint32_t Op = 0b0011'1010'0000'0000'0000'1000'0000'1101;
EvaluateIntoFlags(Op, 1, rn);
}
// Conditional compare - register
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0011'1010'010 << 21;
ConditionalCompare(Op, 0, 0b00, 0, s, rn, rm, flags, Cond);
}
void ccmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0011'1010'010 << 21;
ConditionalCompare(Op, 1, 0b00, 0, s, rn, rm, flags, Cond);
}
// Conditional compare - immediate
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, uint32_t rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
LOGMAN_THROW_A_FMT((rm & ~0b1'1111) == 0, "Comparison imm too large");
constexpr uint32_t Op = 0b0011'1010'010 << 21;
ConditionalCompare(Op, 0, 0b10, 0, s, rn, rm, flags, Cond);
}
void ccmp(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, uint32_t rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
LOGMAN_THROW_A_FMT((rm & ~0b1'1111) == 0, "Comparison imm too large");
constexpr uint32_t Op = 0b0011'1010'010 << 21;
ConditionalCompare(Op, 1, 0b10, 0, s, rn, rm, flags, Cond);
}
// Conditional select
void csel(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0001'1010'100 << 21;
ConditionalCompare(Op, 0, 0b00, s, rd, rn, rm, Cond);
}
void cset(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0001'1010'100 << 21;
ConditionalCompare(Op, 0, 0b01, s, rd, FEXCore::ARMEmitter::Reg::zr, FEXCore::ARMEmitter::Reg::zr, static_cast<FEXCore::ARMEmitter::Condition>(FEXCore::ToUnderlying(Cond) ^ FEXCore::ToUnderlying(FEXCore::ARMEmitter::Condition::CC_NE)));
}
void csinc(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0001'1010'100 << 21;
ConditionalCompare(Op, 0, 0b01, s, rd, rn, rm, Cond);
}
void csinv(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0001'1010'100 << 21;
ConditionalCompare(Op, 1, 0b00, s, rd, rn, rm, Cond);
}
void csneg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0001'1010'100 << 21;
ConditionalCompare(Op, 1, 0b01, s, rd, rn, rm, Cond);
}
// Data processing - 3 source
void madd(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
constexpr uint32_t Op = 0b001'1011'000U << 21;
DataProcessing_3Source(Op, 0, s, rd, rn, rm, ra);
}
void mul(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
madd(s, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
}
void msub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
constexpr uint32_t Op = 0b001'1011'000U << 21;
DataProcessing_3Source(Op, 1, s, rd, rn, rm, ra);
}
void mneg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
msub(s, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
}
void smaddl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
constexpr uint32_t Op = 0b001'1011'001U << 21;
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
}
void smull(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
smaddl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
}
void smsubl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
constexpr uint32_t Op = 0b001'1011'001U << 21;
DataProcessing_3Source(Op, 1, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
}
void smnegl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
smsubl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
}
void smulh(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
constexpr uint32_t Op = 0b001'1011'010U << 21;
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
}
void umaddl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
constexpr uint32_t Op = 0b001'1011'101U << 21;
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
}
void umull(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
umaddl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
}
void umsubl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
constexpr uint32_t Op = 0b001'1011'101U << 21;
DataProcessing_3Source(Op, 1, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
}
void umnegl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
umsubl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
}
void umulh(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
constexpr uint32_t Op = 0b001'1011'110U << 21;
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
}
private:
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b001'0010'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
}
void ands(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b111'0010'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
}
void orr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b011'0010'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
}
void eor(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b101'0010'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
}
void sbfm(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b0001'0011'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, s == ARMEmitter::Size::i64Bit, immr, imms);
}
void bfm(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b0011'0011'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, s == ARMEmitter::Size::i64Bit, immr, imms);
}
// 4.1.64 - Data processing - Immediate
void DataProcessing_PCRel_Imm(uint32_t Op, FEXCore::ARMEmitter::Register rd, uint32_t Imm) {
// Ensure the immediate is masked.
Imm &= 0b1'1111'1111'1111'1111'1111U;
uint32_t Instr = Op;
Instr |= (Imm & 0b11) << 29;
Instr |= (Imm >> 2) << 5;
Instr |= Encode_rd(rd);
dc32(Instr);
}
void DataProcessing_AddSub_Imm(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12) {
bool TooLarge = (Imm & ~0b1111'1111'1111U) != 0;
if (TooLarge && !LSL12 && ((Imm >> 12) & ~0b1111'1111'1111U) == 0) {
// We can convert an immediate
TooLarge = false;
LSL12 = true;
Imm >>= 12;
}
LOGMAN_THROW_AA_FMT(TooLarge == false, "Imm amount too large: 0x{:x}", Imm);
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= LSL12 << 22;
Instr |= Imm << 10;
Instr |= Encode_rn(rn);
Instr |= Encode_rd(rd);
dc32(Instr);
}
// Move Wide
void DataProcessing_MoveWide(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= Imm << 5;
Instr |= Offset << 21;
Instr |= Encode_rd(rd);
dc32(Instr);
}
// Logical immediate
void DataProcessing_Logical_Imm(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= n << 22;
Instr |= immr << 16;
Instr |= imms << 10;
Instr |= Encode_rn(rn);
Instr |= Encode_rd(rd);
dc32(Instr);
}
void DataProcessing_Extract(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
// Current ARMv8 spec hardcodes SF == N for this class of instructions.
// Anythign else is undefined behaviour.
const uint32_t N = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 22) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= N;
Instr |= Encode_rm(rm);
Instr |= Imm << 10;
Instr |= Encode_rn(rn);
Instr |= Encode_rd(rd);
dc32(Instr);
}
// Data-processing - 2 source
void DataProcessing_2Source(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= Encode_rm(rm);
Instr |= Encode_rn(rn);
Instr |= Encode_rd(rd);
dc32(Instr);
}
// Data processing - 1 source
template<typename T>
void DataProcessing_1Source(uint32_t Op, FEXCore::ARMEmitter::Size s, T rd, T rn) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= Encode_rn(rn);
Instr |= Encode_rd(rd);
dc32(Instr);
}
// AddSub - shifted register
void DataProcessing_Shifted_Reg(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift, uint32_t amt) {
LOGMAN_THROW_AA_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= FEXCore::ToUnderlying(Shift) << 22;
Instr |= Encode_rm(rm);
Instr |= static_cast<uint32_t>(amt) << 10;
Instr |= Encode_rn(rn);
Instr |= Encode_rd(rd);
dc32(Instr);
}
// AddSub - extended register
void DataProcessing_Extended_Reg(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= Encode_rm(rm);
Instr |= FEXCore::ToUnderlying(Option) << 13;
Instr |= static_cast<uint32_t>(Shift) << 10;
Instr |= Encode_rn(rn);
Instr |= Encode_rd(rd);
dc32(Instr);
}
// Conditional compare - register
template<typename T>
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, uint32_t o3, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, T rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= o1 << 30;
Instr |= Encode_rm(rm);
Instr |= FEXCore::ToUnderlying(Cond) << 12;
Instr |= o2 << 10;
Instr |= Encode_rn(rn);
Instr |= o3 << 4;
Instr |= FEXCore::ToUnderlying(flags);
dc32(Instr);
}
template<typename T>
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, T rm, FEXCore::ARMEmitter::Condition Cond) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= o1 << 30;
Instr |= Encode_rm(rm);
Instr |= FEXCore::ToUnderlying(Cond) << 12;
Instr |= o2 << 10;
Instr |= Encode_rn(rn);
Instr |= Encode_rd(rd);
dc32(Instr);
}
// Data-processing - 3 source
void DataProcessing_3Source(uint32_t Op, uint32_t Op0, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= Encode_rm(rm);
Instr |= Op0 << 15;
Instr |= Encode_ra(ra);
Instr |= Encode_rn(rn);
Instr |= Encode_rd(rd);
dc32(Instr);
}
void EvaluateIntoFlags(uint32_t op, uint32_t size, WRegister rn) {
uint32_t Instr = op;
Instr |= size << 14;
Instr |= rn.Idx() << 5;
dc32(Instr);
}
File diff suppressed because it is too large. Load diff
@@ -1,322 +0,0 @@
/* Branch instruction emitters.
*
* Most of these instructions will use `BackwardLabel`, `ForwardLabel`, or `BiDirectionLabel` to determine where a branch targets.
*/
public:
// Branches, Exception Generating and System instructions
public:
// Conditional branch immediate
///< Branch conditional
void b(FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm);
}
void b(FEXCore::ARMEmitter::Condition Cond, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
void b(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
}
void b(FEXCore::ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
b(Cond, &Label->Backward);
}
else {
b(Cond, &Label->Forward);
}
}
///< Branch consistent conditional
void bc(FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm);
}
void bc(FEXCore::ARMEmitter::Condition Cond, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
void bc(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
}
void bc(FEXCore::ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
bc(Cond, &Label->Backward);
}
else {
bc(Cond, &Label->Forward);
}
}
// Unconditional branch register
void br(FEXCore::ARMEmitter::Register rn) {
constexpr uint32_t Op = 0b1101011 << 25 |
0b0'000 << 21 | // opc
0b1'1111 << 16 | // op2
0b0000'00 << 10 | // op3
0b0'0000; // op4
UnconditionalBranch(Op, rn);
}
void blr(FEXCore::ARMEmitter::Register rn) {
constexpr uint32_t Op = 0b1101011 << 25 |
0b0'001 << 21 | // opc
0b1'1111 << 16 | // op2
0b0000'00 << 10 | // op3
0b0'0000; // op4
UnconditionalBranch(Op, rn);
}
void ret(FEXCore::ARMEmitter::Register rn = FEXCore::ARMEmitter::Reg::r30) {
constexpr uint32_t Op = 0b1101011 << 25 |
0b0'010 << 21 | // opc
0b1'1111 << 16 | // op2
0b0000'00 << 10 | // op3
0b0'0000; // op4
UnconditionalBranch(Op, rn);
}
// Unconditional branch immediate
void b(uint32_t Imm) {
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, Imm);
}
void b(BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
}
void b(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
}
void b(BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
b(&Label->Backward);
}
else {
b(&Label->Forward);
}
}
void bl(uint32_t Imm) {
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, Imm);
}
void bl(BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
}
void bl(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
}
void bl(BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
bl(&Label->Backward);
}
else {
bl(&Label->Forward);
}
}
// Compare and branch
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, Imm);
}
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, 0);
}
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
cbz(s, rt, &Label->Backward);
}
else {
cbz(s, rt, &Label->Forward);
}
}
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, Imm);
}
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, 0);
}
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
cbnz(s, rt, &Label->Backward);
}
else {
cbnz(s, rt, &Label->Forward);
}
}
// Test and branch immediate
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, Imm);
}
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, 0);
}
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
tbz(rt, Bit, &Label->Backward);
}
else {
tbz(rt, Bit, &Label->Forward);
}
}
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, Imm);
}
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
}
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
tbnz(rt, Bit, &Label->Backward);
}
else {
tbnz(rt, Bit, &Label->Forward);
}
}
private:
// Conditional branch immediate
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
uint32_t Instr = Op;
Instr |= Op1 << 24;
Instr |= (Imm & 0x7'FFFF) << 5;
Instr |= Op0 << 4;
Instr |= FEXCore::ToUnderlying(Cond);
dc32(Instr);
}
// Unconditional branch register
void UnconditionalBranch(uint32_t Op, FEXCore::ARMEmitter::Register rn) {
uint32_t Instr = Op;
Instr |= Encode_rn(rn);
dc32(Instr);
}
// Unconditional branch - immediate
void UnconditionalBranch(uint32_t Op, uint32_t Imm) {
uint32_t Instr = Op;
Instr |= Imm & 0x3FF'FFFF;
dc32(Instr);
}
// Compare and branch
void CompareAndBranch(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= (Imm & 0x7'FFFF) << 5;
Instr |= Encode_rt(rt);
dc32(Instr);
}
// Test and branch - immediate
void TestAndBranch(uint32_t Op, FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
uint32_t Instr = Op;
Instr |= (Bit >> 5) << 31;
Instr |= (Bit & 0b1'1111) << 19;
Instr |= (Imm & 0x3FFF) << 5;
Instr |= Encode_rt(rt);
dc32(Instr);
}
@@ -1,105 +0,0 @@
#pragma once
#include <cstddef>
#include <cstdint>
#include <cstring>
namespace FEXCore::ARMEmitter {
class Buffer {
public:
Buffer() {
SetBuffer(nullptr, 0);
}
Buffer(uint8_t* Base, uint64_t BaseSize) {
SetBuffer(Base, BaseSize);
}
void SetBuffer(uint8_t* Base, uint64_t BaseSize) {
BufferBase = Base;
CurrentOffset = BufferBase;
Size = BaseSize;
}
void dc8(uint8_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc16(uint16_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc32(uint32_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc64(uint64_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void EmitString(const char *String) {
const auto StringLength = strlen(String);
memcpy(CurrentOffset, String, StringLength);
CurrentOffset += StringLength;
}
void Align() {
// Align the buffer to instruction size
auto CurrentAlignment = reinterpret_cast<uint64_t>(CurrentOffset) & 0b11;
if (!CurrentAlignment) {
return;
}
CurrentOffset += 4 - CurrentAlignment;
}
template<typename T>
T GetCursorAddress() const {
return reinterpret_cast<T>(CurrentOffset);
}
static void ClearICache(void* Begin, std::size_t Length) {
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
}
size_t GetCursorOffset() const {
return static_cast<size_t>(CurrentOffset - BufferBase);
}
uint8_t *GetBufferBase() const {
return BufferBase;
}
void CursorIncrement(size_t Size) {
CurrentOffset += Size;
}
void SetCursorOffset(size_t Offset) {
CurrentOffset = BufferBase + Offset;
}
uint64_t GetBufferSize() const {
return Size;
}
template<typename T>
size_t GetCursorOffsetFromAddress(const T* Address) const {
return static_cast<size_t>(reinterpret_cast<const uint8_t*>(Address) - BufferBase);
}
protected:
void ResetBuffer() {
CurrentOffset = BufferBase;
}
uint8_t* BufferBase;
uint8_t* CurrentOffset;
uint64_t Size;
};
}
@@ -1,771 +0,0 @@
#pragma once
#include "Interface/Core/ArchHelpers/CodeEmitter/Buffer.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <aarch64/assembler-aarch64.h>
#include <cstdint>
#include <utility>
#include <type_traits>
#include <vector>
/*
* Welcome to FEX-Emu's custom AArch64 emitter.
* This was written specifically to avoid the performance cost of the vixl emitter.
*
* There are some specific design constraints in this design to target a couple features:
* - High performance
* - Low CPU cache performance hit
* - Significantly reduced code footprint
* - Low number of branches
*
* These requirements are mostly achieved by removing a bunch of developer conveniences
* that vixl provides. The developer needs to take a lot of care to not shoot themselves in the foot.
*
* Misc design decisions:
* - Registers are encoded as basic uint32_t enums.
* - Converting between different registers is zero-cost.
* - Passing around as arguments are as cheap as registers
* - Contrast to vixl where every register requires living on the stack.
* - Registers can get encoded in to instructions with a simple `BFM` instruction.
*
* - Instructions are very simply emitted, allowing direct inlining most of the time.
* - These are simple enough that multiple back-to-back instructions get optimized to 128-bit load-store operations.
* - Contrast to vixl where pretty much no instruction emitter gets inlined.
*
* - Instruction emitters are /mostly/ unsized. Most instructions take a size argument first, which gets encoded
* directly in to the instruction.
* - Contrast to vixl where the register arguments are how the instructions determine operating size.
* - Size argument allows FEX to use `CSEL` to select a size at runtime, instead of branching.
* - Some instructions are explicitly sized based on register type. Read comments in the respective `inl` files to
* see why.
* Some scalar/vector operations are an example of this.
*
* - Almost zero helper functions.
* - Primary exception to this rule is load-store operations. These will use a helper to make
* it easier to select the correct load-store instruction. Mostly because these are a nightmare selecting
* the right instruction.
*/
namespace FEXCore::ARMEmitter {
/*
* This `Size` enum is used for most ALU operations.
* These follow the AArch64 encoding style in most cases.
*/
enum class Size : uint32_t {
i32Bit = 0,
i64Bit,
};
// This allows us to get the `Size` enum in bits.
template<Size size>
constexpr size_t RegSizeInBits() {
constexpr size_t RegSize[] = {
32, 64, 128,
};
return RegSize[FEXCore::ToUnderlying(size)];
}
[[maybe_unused]]
static inline size_t RegSizeInBits(Size size) {
constexpr size_t RegSize[] = {
32, 64, 128,
};
return RegSize[FEXCore::ToUnderlying(size)];
}
/* This `SubRegSize` enum is used for most ASIMD operations.
* These follow the AArch64 encoding style in most cases.
*/
enum class SubRegSize : uint32_t {
i8Bit = 0b00,
i16Bit = 0b01,
i32Bit = 0b10,
i64Bit = 0b11,
i128Bit = 0b100,
};
// This allows us to get the `SubRegSize` in bits.
template<SubRegSize size>
constexpr size_t SubRegSizeInBits() {
return (1 << FEXCore::ToUnderlying(size)) * 8;
}
[[maybe_unused]]
static inline size_t SubRegSizeInBits(SubRegSize size) {
return (1 << FEXCore::ToUnderlying(size)) * 8;
}
/* This `ScalarRegSize` enum is used for most scalar float
* operations.
*
* This is specifically duplicated from `SubRegSize` to have strongly
* typed functions.
*
* `ScalarRegSize` specifically doesn't have `i128Bit` because scalar operations
* can't operate at 128-bit.
*/
enum class ScalarRegSize : uint32_t {
i8Bit = 0b00,
i16Bit = 0b01,
i32Bit = 0b10,
i64Bit = 0b11,
};
// This allows us to get the `ScalarRegSize` in bits.
template<ScalarRegSize size>
constexpr size_t ScalarRegSizeInBits() {
return (1 << FEXCore::ToUnderlying(size)) * 8;
}
[[maybe_unused]]
static inline size_t ScalarRegSizeInBits(ScalarRegSize size) {
return (1 << FEXCore::ToUnderlying(size)) * 8;
}
/* This `VectorRegSizePair` union allows us to have an overlapping type
* to select a scalar operation or a vector depending on which operation
* we pass in.
* Useful in FEX's vector operations that behave as scalar or vector
* depending on various factors. But since the operation will have the sa,e
* element size, we want to choose the operation more easily
*/
union VectorRegSizePair {
ScalarRegSize Scalar;
SubRegSize Vector;
};
// This allows us to create a `VectorRegSizePair` union.
[[maybe_unused]]
static inline VectorRegSizePair ToVectorSizePair(SubRegSize size) {
return VectorRegSizePair {.Vector = size};
}
[[maybe_unused]]
static inline VectorRegSizePair ToVectorSizePair(ScalarRegSize size) {
return VectorRegSizePair {.Scalar = size};
}
// This `ShiftType` enum is used for ALU shift-register encoded instructions.
enum class ShiftType : uint32_t {
LSL = 0,
LSR,
ASR,
ROR,
};
// This `ExtendedType` enum is used for ALU extended-register encoded instructions.
enum class ExtendedType : uint32_t {
UXTB = 0b000,
UXTH = 0b001,
UXTW = 0b010,
UXTX = 0b011,
SXTB = 0b100,
SXTH = 0b101,
SXTW = 0b110,
SXTX = 0b111,
LSL_32 = UXTW,
LSL_64 = UXTX,
};
// This `Condition` enum is used for various conditional instructions.
enum class Condition : uint32_t {
// Meaning: Int - Float
CC_EQ = 0, // Equal - Equal
CC_NE, // Not Eq - Not Eq or unordered
CC_CS, // Carry set - Greater than, equal, or unordered
CC_CC, // Carry clear - Less than
CC_MI, // Minus/Negative - Less than
CC_PL, // Plus, positive or zero - GT, equal, or unordered
CC_VS, // Overflow - Unordered
CC_VC, // No Overflow - Ordered
CC_HI, // Unsigned higher - GT, or unordered
CC_LS, // Unsigned lower or same - LT or EQ
CC_GE, // Signed GT or EQ - GT or EQ
CC_LT, // Signed LT - LT or Unordered
CC_GT, // Signed GT - GT
CC_LE, // Signed LT or EQ - LT, EQ, or Unordered
CC_AL, // Always - Always
CC_NV, // Always - Always
// Aliases
CC_HS = CC_CS,
CC_LO = CC_CC,
};
/*
* This `StatusFlags` enum is used for conditional compare encoded instructions.
* These directly encode to the `nzcv` flags.
*/
enum class StatusFlags : uint32_t {
None = 0,
Flag_V = 0b0001,
Flag_C = 0b0010,
Flag_Z = 0b0100,
Flag_N = 0b1000,
Flag_NZCV = Flag_N | Flag_Z | Flag_C | Flag_V,
};
/*
* This `IndexType` enum is used for load-store instructions.
* Not all load-store instructions use this, so the user needs to be careful.
*/
enum class IndexType {
POST,
OFFSET,
PRE,
UNPRIVILEGED,
};
/* This `SVEMemOperand` class is used for the helper SVE load-store instructions.
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
*/
class SVEMemOperand final {
public:
SVEMemOperand(XRegister rn, XRegister rm = XReg::zr)
: rn {rn}
, MetaType {
.ScalarScalarType {
.Header = { .MemType = TYPE_SCALAR_SCALAR },
.rm = rm,
}
} {}
SVEMemOperand(XRegister rn, int32_t imm = 0)
: rn {rn}
, MetaType {
.ScalarImmType {
.Header = { .MemType = TYPE_SCALAR_IMM },
.Imm = imm,
}
} {}
Register rn;
enum Type {
TYPE_SCALAR_SCALAR,
TYPE_SCALAR_IMM,
TYPE_SCALAR_VECTOR,
TYPE_VECTOR_IMM,
};
struct HeaderStruct {
Type MemType;
};
union {
HeaderStruct Header;
struct {
HeaderStruct Header;
Register rm;
} ScalarScalarType;
struct {
HeaderStruct Header;
int32_t Imm;
} ScalarImmType;
struct {
HeaderStruct Header;
ZRegister zm;
// TODO: Implement support for modifier
} ScalarVectorType;
struct {
HeaderStruct Header;
// rn will be a ZRegister
int32_t Imm;
} VectorImmType;
} MetaType;
};
/* This `ExtendedMemOperand` class is used for the helper load-store instructions.
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
*/
class ExtendedMemOperand final {
public:
ExtendedMemOperand(XRegister rn, XRegister rm = XReg::zr, ExtendedType Option = ExtendedType::LSL_64, uint32_t Shift = 0)
: rn {rn}
, MetaType {
.ExtendedType {
.Header = { .MemType = TYPE_EXTENDED },
.rm = rm,
.Option = Option,
.Shift = Shift,
}
} {}
ExtendedMemOperand(XRegister rn, IndexType Index = IndexType::OFFSET, int32_t Imm = 0)
: rn {rn}
, MetaType {
.ImmType {
.Header = { .MemType = TYPE_IMM },
.Index = Index,
.Imm = Imm,
}
} {}
Register rn;
enum Type {
TYPE_EXTENDED,
TYPE_IMM,
};
struct HeaderStruct {
Type MemType;
};
union {
HeaderStruct Header;
struct {
HeaderStruct Header;
Register rm;
ExtendedType Option;
uint32_t Shift;
} ExtendedType;
struct {
HeaderStruct Header;
IndexType Index;
int32_t Imm;
} ImmType;
} MetaType;
};
template<uint32_t op0, uint32_t op1, uint32_t CRn, uint32_t CRm, uint32_t op2>
constexpr uint32_t GenSystemReg() {
return op0 << 19 |
op1 << 16 |
CRn << 12 |
CRm << 8 |
op2 << 5;
};
// This `SystemRegister` enum is used for the mrs/msr instructions.
enum class SystemRegister : uint32_t {
CTR_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b001>(),
DCZID_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b111>(),
TPIDR_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b010>(),
RNDR = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b000>(),
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>(),
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>(),
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>(),
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>(),
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>(),
};
template<uint32_t op1, uint32_t CRm, uint32_t op2>
constexpr uint32_t GenDCReg() {
return op1 << 16 |
CRm << 8 |
op2 << 5;
};
// This `DataCacheOperation` enum is used for the dc instruction.
enum class DataCacheOperation : uint32_t {
IVAC = GenDCReg<0b000, 0b0110, 0b001>(),
ISW = GenDCReg<0b000, 0b0110, 0b010>(),
CSW = GenDCReg<0b000, 0b1010, 0b010>(),
CISW = GenDCReg<0b000, 0b1110, 0b010>(),
ZVA = GenDCReg<0b011, 0b0100, 0b001>(),
CVAC = GenDCReg<0b011, 0b1010, 0b001>(),
CVAU = GenDCReg<0b011, 0b1011, 0b001>(),
CIVAC = GenDCReg<0b011, 0b1110, 0b001>(),
// MTE2
IGVAC = GenDCReg<0b000, 0b0110, 0b011>(),
IGSW = GenDCReg<0b000, 0b0110, 0b100>(),
IGDVAC = GenDCReg<0b000, 0b0110, 0b101>(),
IGDSW = GenDCReg<0b000, 0b0110, 0b110>(),
CGSW = GenDCReg<0b000, 0b1010, 0b100>(),
CGDSW = GenDCReg<0b000, 0b1010, 0b110>(),
CIGSW = GenDCReg<0b000, 0b1110, 0b100>(),
CIGDSW = GenDCReg<0b000, 0b1110, 0b110>(),
// MTE
GVA = GenDCReg<0b011, 0b0100, 0b011>(),
GZVA = GenDCReg<0b011, 0b0100, 0b100>(),
CGVAC = GenDCReg<0b011, 0b1010, 0b011>(),
CGDVAC = GenDCReg<0b011, 0b1010, 0b101>(),
CGVAP = GenDCReg<0b011, 0b1100, 0b011>(),
CGDVAP = GenDCReg<0b011, 0b1100, 0b101>(),
CGVADP = GenDCReg<0b011, 0b1101, 0b011>(),
CGDVADP = GenDCReg<0b011, 0b1101, 0b101>(),
CIGVAC = GenDCReg<0b011, 0b1110, 0b011>(),
CIGDVAC = GenDCReg<0b011, 0b1110, 0b101>(),
// DPB
CVAP = GenDCReg<0b011, 0b1100, 0b001>(),
// DPB2
CVADP = GenDCReg<0b011, 0b1101, 0b001>(),
};
template<uint32_t CRm, uint32_t op2>
constexpr uint32_t GenHintBarrierReg() {
return CRm << 8 |
op2 << 5;
}
// This `HintRegister` enum is used for the hint instruction.
enum class HintRegister : uint32_t {
NOP = GenHintBarrierReg<0b0000, 0b000>(),
YIELD = GenHintBarrierReg<0b0000, 0b001>(),
WFE = GenHintBarrierReg<0b0000, 0b010>(),
WFI = GenHintBarrierReg<0b0000, 0b011>(),
SEV = GenHintBarrierReg<0b0000, 0b100>(),
SEVL = GenHintBarrierReg<0b0000, 0b101>(),
DGH = GenHintBarrierReg<0b0000, 0b110>(),
CSDB = GenHintBarrierReg<0b0010, 0b100>(),
};
// This `BarrierRegister` enum is used for the various barrier instructions.
enum class BarrierRegister : uint32_t {
CLREX = GenHintBarrierReg<0b0000, 0b010>(),
TCOMMIT = GenHintBarrierReg<0b0000, 0b011>(),
DSB = GenHintBarrierReg<0b0000, 0b100>(),
DMB = GenHintBarrierReg<0b0000, 0b101>(),
ISB = GenHintBarrierReg<0b0000, 0b110>(),
SB = GenHintBarrierReg<0b0000, 0b111>(),
};
// This `BarrierScope` enum is used for the dsb/dmb instructions.
enum class BarrierScope : uint32_t {
// Outer shareable
OSHLD = 0b0001,
OSHST = 0b0010,
OSH = 0b0011,
// Non shareable
NSHLD = 0b0101,
NSHST = 0b0110,
NSH = 0b0111,
// Inner shareable
ISHLD = 0b1001,
ISHST = 0b1010,
ISH = 0b1011,
// Full System visibility
LD = 0b1101,
ST = 0b1110,
SY = 0b1111,
};
// This `Prefetch` enum is used for prefetch instructions.
enum class Prefetch : uint32_t {
// Prefetch for load
PLDL1KEEP = 0b00000,
PLDL1STRM = 0b00001,
PLDL2KEEP = 0b00010,
PLDL2STRM = 0b00011,
PLDL3KEEP = 0b00100,
PLDL3STRM = 0b00101,
// Preload instructions
PLIL1KEEP = 0b01000,
PLIL1STRM = 0b01001,
PLIL2KEEP = 0b01010,
PLIL2STRM = 0b01011,
PLIL3KEEP = 0b01100,
PLIL3STRM = 0b01101,
// Preload for store
PSTL1KEEP = 0b10000,
PSTL1STRM = 0b10001,
PSTL2KEEP = 0b10010,
PSTL2STRM = 0b10011,
PSTL3KEEP = 0b10100,
PSTL3STRM = 0b10101,
};
// This `PredicatePattern` enun is used for some SVE instructions.
enum class PredicatePattern : uint32_t {
SVE_POW2 = 0b00000,
SVE_VL1 = 0b00001,
SVE_VL2 = 0b00010,
SVE_VL3 = 0b00011,
SVE_VL4 = 0b00100,
SVE_VL5 = 0b00101,
SVE_VL6 = 0b00110,
SVE_VL7 = 0b00111,
SVE_VL8 = 0b01000,
SVE_VL16 = 0b01001,
SVE_VL32 = 0b01010,
SVE_VL64 = 0b01011,
SVE_VL128 = 0b01100,
SVE_VL256 = 0b01101,
SVE_MUL4 = 0b11101,
SVE_MUL3 = 0b11110,
SVE_ALL = 0b11111,
};
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is logically `below` an instruction that uses it.
* Which means that a branch would jump backwards.
*/
struct BackwardLabel {
uint8_t *Location{};
};
/* This `ForwardLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is logically `above` an instruction that uses it.
* Which means that a branch would jump forwards.
*
* This can be bound to multiple instructions, so it needs a vector for each bind instruction type.
*/
struct ForwardLabel {
struct Instructions {
enum class InstType {
ADR,
ADRP,
B,
BC,
TEST_BRANCH,
RELATIVE_LOAD,
LONG_ADDRESS_GEN,
};
uint8_t *Location{};
InstType Type;
};
std::vector<Instructions> Insts{};
};
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is in either direction of an instruction that uses it.
* Which means a branch could jump backwards or forwards depending on situation.
*/
struct BiDirectionalLabel {
BackwardLabel Backward;
ForwardLabel Forward;
};
// Some FCMA ASIMD instructions support a rotation argument.
enum class Rotation : uint32_t {
ROTATE_0 = 0b00,
ROTATE_90 = 0b01,
ROTATE_180 = 0b10,
ROTATE_270 = 0b11,
};
// This is an emitter that is designed around the smallest code bloat as possible.
// Eschewing most developer convenience in order to keep code as small as possible.
// Choices:
// - Size of ops passed as an argument rather than template to let the compiler use csel instead of branching.
// - Registers are unsized so they can be passed in a GPR and not need conversion operations
class Emitter : public FEXCore::ARMEmitter::Buffer {
public:
Emitter() = default;
Emitter(uint8_t* Base, uint64_t BaseSize)
: Buffer (Base, BaseSize) {
}
// Bind a backward label to an address.
// Address that is bound is the current emitter location.
void Bind(BackwardLabel *Label) {
LOGMAN_THROW_AA_FMT(Label->Location == nullptr, "Trying to bind a label twice");
Label->Location = GetCursorAddress<uint8_t*>();
}
// Bind a forward label to a location.
// This walks all the instructions in the label's vector.
// Then backpatching all instructions that have used the label.
template<bool WarnAboutEmpty = false>
void Bind(ForwardLabel *Label) {
if constexpr (WarnAboutEmpty) {
LOGMAN_THROW_A_FMT(Label->Insts.empty() == false, "Binding forward label that didn't have any instructions using it");
}
uint8_t *CurrentAddress = GetCursorAddress<uint8_t*>();
for (const auto &Inst : Label->Insts) {
// Patch up the instructions
switch (Inst.Type) {
case ForwardLabel::Instructions::InstType::ADR: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::ADRP: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::B: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= Offset;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::TEST_BRANCH: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::BC:
case ForwardLabel::Instructions::InstType::RELATIVE_LOAD: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN: {
uint32_t *Instructions = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
auto OriginalOffset = GetCursorOffset();
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
SetCursorOffset(InstOffset);
// We encoded the destination register in to the first instruction space.
// Read it back.
ARMEmitter::Register DestReg(Instructions[0]);
if (IsADRRange(ImmInstTwo)) {
// If within ADR range from the second instruction, then we can emit NOP+ADR
nop();
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
}
else if (IsADRPRange(ImmInstOne)) {
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
// First check if we are in non-offset range for second instruction.
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
// We can emit nop + adrp
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
}
else {
// Not aligned, need adrp + add
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
}
}
else {
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
FEX_UNREACHABLE;
}
SetCursorOffset(OriginalOffset);
break;
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
}
}
// Bind a bidirectional location to a location.
// Binds both forwards and backwards depending on how the label was used.
void Bind(BiDirectionalLabel *Label) {
if (!Label->Backward.Location) {
Bind(&Label->Backward);
}
Bind<false>(&Label->Forward);
}
public:
// TODO: Implement SME when it matters.
#include "Interface/Core/ArchHelpers/CodeEmitter/ALUOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/BranchOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/LoadstoreOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/SystemOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/ScalarOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/ASIMDOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/SVEOps.inl"
private:
template<typename T>
uint32_t Encode_ra(T Reg) const {
return Reg.Idx() << 10;
}
uint32_t Encode_ra(uint32_t Reg) const {
return Reg << 10;
}
template<typename T>
uint32_t Encode_rt2(T Reg) const {
return Reg.Idx() << 10;
}
template<>
uint32_t Encode_rt2(uint32_t Reg) const {
return Reg << 10;
}
template<typename T>
uint32_t Encode_rm(T Reg) const {
return Reg.Idx() << 16;
}
uint32_t Encode_rm(uint32_t Reg) const {
return Reg << 16;
}
template<typename T>
uint32_t Encode_rs(T Reg) const {
return Reg.Idx() << 16;
}
uint32_t Encode_rs(uint32_t Reg) const {
return Reg << 16;
}
template<typename T>
uint32_t Encode_rn(T Reg) const {
return Reg.Idx() << 5;
}
uint32_t Encode_rn(uint32_t Reg) const {
return Reg << 5;
}
template<typename T>
uint32_t Encode_rd(T Reg) const {
return Reg.Idx();
}
uint32_t Encode_rd(uint32_t Reg) const {
return Reg;
}
template<typename T>
uint32_t Encode_rt(T Reg) const {
return Reg.Idx();
}
template<>
uint32_t Encode_rt(Prefetch Reg) const {
return FEXCore::ToUnderlying(Reg);
}
uint32_t Encode_rt(uint32_t Reg) const {
return Reg;
}
template<typename T>
uint32_t Encode_pd(T Reg) const {
return FEXCore::ToUnderlying(Reg);
}
};
}
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -1,175 +0,0 @@
/* System instruction emitters.
*
* This is mostly a mashup of various instruction types.
* Nothing follows an explicit pattern since they are mostly different.
*/
public:
// System with result
// TODO: SYSL
// System Instruction
// TODO: AT
// TODO: CFP
// TODO: CPP
void dc(FEXCore::ARMEmitter::DataCacheOperation DCOp, FEXCore::ARMEmitter::Register rt) {
constexpr uint32_t Op = 0b1101'0101'0000'1000'0111 << 12;
SystemInstruction(Op, 0, FEXCore::ToUnderlying(DCOp), rt);
}
// TODO: DVP
// TODO: IC
// TODO: TLBI
// Exception generation
void svc(uint32_t Imm) {
ExceptionGeneration(0b000, 0b000, 0b01, Imm);
}
void hvc(uint32_t Imm) {
ExceptionGeneration(0b000, 0b000, 0b10, Imm);
}
void smc(uint32_t Imm) {
ExceptionGeneration(0b000, 0b000, 0b11, Imm);
}
void brk(uint32_t Imm) {
ExceptionGeneration(0b001, 0b000, 0b00, Imm);
}
void hlt(uint32_t Imm) {
ExceptionGeneration(0b010, 0b000, 0b00, Imm);
}
void tcancel(uint32_t Imm) {
ExceptionGeneration(0b011, 0b000, 0b00, Imm);
}
void dcps1(uint32_t Imm) {
ExceptionGeneration(0b101, 0b000, 0b01, Imm);
}
void dcps2(uint32_t Imm) {
ExceptionGeneration(0b101, 0b000, 0b10, Imm);
}
void dcps3(uint32_t Imm) {
ExceptionGeneration(0b101, 0b000, 0b11, Imm);
}
// System instructions with register argument
void wfet(FEXCore::ARMEmitter::Register rt) {
SystemInstructionWithReg(0b0000, 0b000, rt);
}
void wfit(FEXCore::ARMEmitter::Register rt) {
SystemInstructionWithReg(0b0000, 0b001, rt);
}
// Hints
void nop() {
Hint(FEXCore::ARMEmitter::HintRegister::NOP);
}
void yield() {
Hint(FEXCore::ARMEmitter::HintRegister::YIELD);
}
void wfe() {
Hint(FEXCore::ARMEmitter::HintRegister::WFE);
}
void wfi() {
Hint(FEXCore::ARMEmitter::HintRegister::WFI);
}
void sev() {
Hint(FEXCore::ARMEmitter::HintRegister::SEV);
}
void sevl() {
Hint(FEXCore::ARMEmitter::HintRegister::SEVL);
}
void dgh() {
Hint(FEXCore::ARMEmitter::HintRegister::DGH);
}
void csdb() {
Hint(FEXCore::ARMEmitter::HintRegister::CSDB);
}
// Barriers
void clrex(uint32_t imm = 15) {
LOGMAN_THROW_AA_FMT(imm < 16, "Immediate out of range");
Barrier(FEXCore::ARMEmitter::BarrierRegister::CLREX, imm);
}
void dsb(FEXCore::ARMEmitter::BarrierScope Scope) {
Barrier(FEXCore::ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
}
void dmb(FEXCore::ARMEmitter::BarrierScope Scope) {
Barrier(FEXCore::ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
}
void isb() {
Barrier(FEXCore::ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(FEXCore::ARMEmitter::BarrierScope::SY));
}
void sb() {
Barrier(FEXCore::ARMEmitter::BarrierRegister::SB, 0);
}
void tcommit() {
Barrier(FEXCore::ARMEmitter::BarrierRegister::TCOMMIT, 0);
}
// System register move
void msr(FEXCore::ARMEmitter::SystemRegister reg, FEXCore::ARMEmitter::Register rt) {
constexpr uint32_t Op = 0b1101'0101'0001 << 20;
SystemRegisterMove(Op, rt, reg);
}
void mrs(FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::SystemRegister reg) {
constexpr uint32_t Op = 0b1101'0101'0011 << 20;
SystemRegisterMove(Op, rd, reg);
}
private:
// Exception Generation
void ExceptionGeneration(uint32_t opc, uint32_t op2, uint32_t LL, uint32_t Imm) {
LOGMAN_THROW_AA_FMT((Imm & 0xFFFF'0000) == 0, "Imm amount too large");
uint32_t Instr = 0b1101'0100 << 24;
Instr |= opc << 21;
Instr |= Imm << 5;
Instr |= op2 << 2;
Instr |= LL;
dc32(Instr);
}
// System instructions with register argument
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, FEXCore::ARMEmitter::Register rt) {
uint32_t Instr = 0b1101'0101'0000'0011'0001 << 12;
Instr |= CRm << 8;
Instr |= op2 << 5;
Instr |= Encode_rt(rt);
dc32(Instr);
}
// Hints
void Hint(FEXCore::ARMEmitter::HintRegister Reg) {
uint32_t Instr = 0b1101'0101'0000'0011'0010'0000'0001'1111U;
Instr |= FEXCore::ToUnderlying(Reg);
dc32(Instr);
}
// Barriers
void Barrier(FEXCore::ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
uint32_t Instr = 0b1101'0101'0000'0011'0011'0000'0001'1111U;
Instr |= CRm << 8;
Instr |= FEXCore::ToUnderlying(Reg);
dc32(Instr);
}
// System Instruction
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, FEXCore::ARMEmitter::Register rt) {
uint32_t Instr = Op;
Instr |= L << 21;
Instr |= SubOp;
Instr |= Encode_rt(rt);
dc32(Instr);
}
// System register move
void SystemRegisterMove(uint32_t Op, FEXCore::ARMEmitter::Register rt, FEXCore::ARMEmitter::SystemRegister reg) {
uint32_t Instr = Op;
Instr |= FEXCore::ToUnderlying(reg);
Instr |= Encode_rt(rt);
dc32(Instr);
}
+22 -168
View File
@@ -3,7 +3,6 @@
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/UContext.h>
#include <FEXCore/Core/X86Enums.h>
#include <signal.h>
#include <string.h>
@@ -11,48 +10,24 @@
#include <stdint.h>
#include <type_traits>
namespace FEXCore::ArchHelpers::Context {
enum ContextFlags : uint32_t {
CONTEXT_FLAG_INJIT = (1U << 0),
CONTEXT_FLAG_32BIT = (1U << 1),
};
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
constexpr uint64_t STACK_COOKIE_MAGIC = 0x4142434445464748ULL;
#endif
struct X86ContextBackup {
// Host State
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
// During debug builds, insert a cookie on the stack.
// This is useful for validation that the stack is trying to be restored from the correct location.
// During stack restore, we ensure this is set to the value we expect.
// If given an incorrect stack location, or corrupted stack then this cookie will be wrong.
uint64_t StackCookie;
#endif
// RIP and RSP is stored in GPRs here
uint64_t GPRs[23];
FEXCore::x86_64::_libc_fpstate FPRState;
uint64_t sa_mask;
bool FaultToTopAndGeneratedException;
// Guest state
int Signal;
uint32_t Flags;
uint64_t OriginalRIP;
uint64_t FPStateLocation;
uint64_t UContextLocation;
uint64_t SigInfoLocation;
FEXCore::Core::CPUState GuestState;
static constexpr int RedZoneSize = 128;
};
struct ArmContextBackup {
// Host State
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
uint64_t StackCookie;
#endif
uint64_t GPRs[31];
uint64_t PrevSP;
uint64_t PrevPC;
@@ -60,27 +35,15 @@ struct ArmContextBackup {
uint32_t FPSR;
uint32_t FPCR;
__uint128_t FPRs[32];
uint64_t sa_mask;
bool FaultToTopAndGeneratedException;
// Guest state
int Signal;
uint32_t Flags;
uint64_t OriginalRIP;
uint64_t FPStateLocation;
uint64_t UContextLocation;
uint64_t SigInfoLocation;
FEXCore::Core::CPUState GuestState;
// Arm64 doesn't have a red zone
static constexpr int RedZoneSize = 0;
};
static inline ucontext_t* GetUContext(void* ucontext) {
ucontext_t* _context = (ucontext_t*)ucontext;
return _context;
}
static inline mcontext_t* GetMContext(void* ucontext) {
ucontext_t* _context = (ucontext_t*)ucontext;
return &_context->uc_mcontext;
@@ -89,26 +52,6 @@ static inline mcontext_t* GetMContext(void* ucontext) {
#ifdef _M_ARM_64
constexpr uint32_t FPR_MAGIC = 0x46508001U;
constexpr uint32_t ESR1_MAGIC = 0x45535201U;
struct HostCTXHeader {
uint32_t Magic;
uint32_t Size;
};
struct HostFPRState {
HostCTXHeader Head;
uint32_t FPSR;
uint32_t FPCR;
__uint128_t FPRs[32];
};
struct HostESRState {
HostCTXHeader Head;
uint64_t ESR;
};
static inline uint64_t GetSp(void* ucontext) {
return GetMContext(ucontext)->sp;
}
@@ -141,74 +84,24 @@ static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
GetMContext(ucontext)->regs[id] = val;
}
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
auto MContext = GetMContext(ucontext);
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&MContext->__reserved[0]);
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
constexpr uint32_t FPR_MAGIC = 0x46508001U;
return HostState->FPRs[id];
}
struct HostCTXHeader {
uint32_t Magic;
uint32_t Size;
};
static inline uint64_t GetArmESR(void* ucontext) {
auto MContext = GetMContext(ucontext);
size_t i = 0;
auto HostState = reinterpret_cast<HostCTXHeader*>(&MContext->__reserved[i]);
do {
if (HostState->Magic == ESR1_MAGIC) {
auto ESR = reinterpret_cast<HostESRState*>(HostState);
return ESR->ESR;
}
i += HostState->Size;
HostState = reinterpret_cast<HostCTXHeader*>(&MContext->__reserved[i]);
} while (HostState->Size != 0);
return 0;
}
constexpr static uint64_t ESR1_EC = 0b111111U << 26;
constexpr static uint64_t ESR1_EC_DataAbort = 0b100100U << 26;
// Write-Not-Read flag
// When set - Abort is due to a write
constexpr static uint64_t ESR1_WNR = 1 << 6;
// DFSC - Default Status Code
// Translation fault - No page mapped
// Permissions fault - Page mapped but with incorrect permission from access.
constexpr static uint64_t ESR1_DataAbort_DFSC = 0b111111;
constexpr static uint64_t ESR1_DataAbort_TranslationFault_EL0 = 0b000111;
constexpr static uint64_t ESR1_DataAbort_PermissionFault_EL0 = 0b001111;
constexpr static uint64_t ESR1_DataAbort_Level = 0b11;
constexpr static uint64_t ESR1_DataAbort_Level_EL3 = 0b00;
constexpr static uint64_t ESR1_DataAbort_Level_EL2 = 0b01;
constexpr static uint64_t ESR1_DataAbort_Level_EL1 = 0b10;
constexpr static uint64_t ESR1_DataAbort_Level_EL0 = 0b11;
static inline uint32_t GetProtectFlags(void* ucontext) {
uint64_t ESR = GetArmESR(ucontext);
LOGMAN_THROW_A_FMT((ESR & ESR1_EC) == ESR1_EC_DataAbort, "Unknown ESR1 EC type: 0x{:x} != 0x{:x}", ESR & ESR1_EC, ESR1_EC_DataAbort);
uint32_t ProtectFlags{};
if ((ESR & ESR1_DataAbort_Level) == ESR1_DataAbort_Level_EL0) {
// Always a user error for us.
ProtectFlags |= X86State::X86_PF_USER;
}
if (ESR & ESR1_WNR) {
// Fault was due to a write
ProtectFlags |= X86State::X86_PF_WRITE;
}
// PF_PROT is not returned to user on x86, so don't return the difference between permission fault and translation fault.
return ProtectFlags;
}
struct HostFPRState {
HostCTXHeader Head;
uint32_t FPSR;
uint32_t FPCR;
__uint128_t FPRs[32];
};
using ContextBackup = ArmContextBackup;
template <typename T>
static inline void BackupContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, ArmContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
memcpy(&Backup->GPRs[0], &_mcontext->regs[0], 31 * sizeof(uint64_t));
@@ -218,31 +111,22 @@ static inline void BackupContext(void* ucontext, T *Backup) {
// Host FPR state starts at _mcontext->reserved[0];
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
LogMan::Throw::A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
Backup->FPSR = HostState->FPSR;
Backup->FPCR = HostState->FPCR;
memcpy(&Backup->FPRs[0], &HostState->FPRs[0], 32 * sizeof(__uint128_t));
// Save the signal mask so we can restore it
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
Backup->StackCookie = STACK_COOKIE_MAGIC;
#endif
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
template <typename T>
static inline void RestoreContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, ArmContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
LogMan::Throw::A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
memcpy(&HostState->FPRs[0], &Backup->FPRs[0], 32 * sizeof(__uint128_t));
HostState->FPCR = Backup->FPCR;
HostState->FPSR = Backup->FPSR;
@@ -252,14 +136,8 @@ static inline void RestoreContext(void* ucontext, T *Backup) {
ArchHelpers::Context::SetPc(ucontext, Backup->PrevPC);
ArchHelpers::Context::SetSp(ucontext, Backup->PrevSP);
memcpy(&_mcontext->regs[0], &Backup->GPRs[0], 31 * sizeof(uint64_t));
// Restore the signal mask now
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
LOGMAN_THROW_A_FMT(Backup->StackCookie == STACK_COOKIE_MAGIC, "Stack cookie didn't match! 0x{:x}", Backup->StackCookie);
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
@@ -292,26 +170,17 @@ static inline void SetState(void* ucontext, uint64_t val) {
}
static inline uint64_t GetArmReg(void* ucontext, uint32_t id) {
ERROR_AND_DIE_FMT("Not impelented for x86 host");
ERROR_AND_DIE("Not impelented for x86 host");
}
static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
ERROR_AND_DIE_FMT("Not impelented for x86 host");
}
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
ERROR_AND_DIE_FMT("Not implemented for x86 host");
}
static inline uint32_t GetProtectFlags(void* ucontext) {
return GetMContext(ucontext)->gregs[REG_ERR];
ERROR_AND_DIE("Not impelented for x86 host");
}
using ContextBackup = X86ContextBackup;
template <typename T>
static inline void BackupContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, X86ContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
// Copy the GPRs
@@ -319,40 +188,25 @@ static inline void BackupContext(void* ucontext, T *Backup) {
// Copy the FPRState
memcpy(&Backup->FPRState, _mcontext->fpregs, sizeof(X86ContextBackup::FPRState));
// XXX: Save 256bit and 512bit AVX register state
// Save the signal mask so we can restore it
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
Backup->StackCookie = STACK_COOKIE_MAGIC;
#endif
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
template <typename T>
static inline void RestoreContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, X86ContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
// Copy the GPRs
memcpy(&_mcontext->gregs[0], &Backup->GPRs[0], sizeof(X86ContextBackup::GPRs));
// Copy the FPRState
memcpy(_mcontext->fpregs, &Backup->FPRState, sizeof(X86ContextBackup::FPRState));
// Restore the signal mask now
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
LOGMAN_THROW_A_FMT(Backup->StackCookie == STACK_COOKIE_MAGIC, "Stack cookie didn't match! 0x{:x}", Backup->StackCookie);
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
#endif
} // namespace FEXCore::ArchHelpers::Context
} // namespace FEXCore::ArchHelpers::Context
@@ -2,7 +2,6 @@
#include <FEXCore/Utils/LogManager.h>
#include <cstring>
#include <fstream>
#include <utility>
namespace FEXCore {
void BlockSamplingData::DumpBlockData() {
@@ -27,7 +26,7 @@ namespace FEXCore {
<< std::endl;
}
Output.close();
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
LogMan::Msg::D("Dumped %d blocks of sampling data", SamplingMap.size());
}
BlockSamplingData::BlockData *BlockSamplingData::GetBlockData(uint64_t RIP) {
+1 -2
View File
@@ -1,7 +1,6 @@
#pragma once
#include <cstdint>
#include <unordered_map>
#include <stdint.h>
namespace FEXCore {
class BlockSamplingData {
-87
View File
@@ -1,87 +0,0 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include <FEXCore/Core/CPUBackend.h>
namespace FEXCore {
namespace CPU {
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {}
CPUBackend::~CPUBackend() {
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
}
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
if (CodeBuffers.empty()) {
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
} else {
if (CodeBuffers.size() > 1) {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
for (size_t i = 1; i < CodeBuffers.size(); i++) {
FreeCodeBuffer(CodeBuffers[i]);
}
CodeBuffers.resize(1);
}
// Set the current code buffer to the initial
CurrentCodeBuffer = &CodeBuffers[0];
if (CurrentCodeBuffer->Size != MaxCodeSize) {
FreeCodeBuffer(*CurrentCodeBuffer);
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer->Size *= 1.5;
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
}
}
} else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
}
return CurrentCodeBuffer;
}
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t *>(
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
if (ThreadState->CTX->Config.GlobalJITNaming()) {
ThreadState->CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
for (auto &Buffer: CodeBuffers) {
auto start = (uintptr_t)Buffer.Ptr;
auto end = start + Buffer.Size;
if (Address >= start && Address < end) {
return true;
}
}
return false;
}
}
}
File diff suppressed because it is too large. Load diff
+32 -230
View File
@@ -1,21 +1,15 @@
#pragma once
#include <cstdint>
#include <functional>
#include <unordered_map>
#include <utility>
#include <vector>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
namespace FEXCore {
namespace Context {
struct Context;
}
// Debugging define to switch what family of CPU we execute as.
// Might be useful if an application makes an assumption about a CPU.
// #define CPUID_AMD
class CPUIDEmu final {
private:
constexpr static uint32_t CPUID_VENDOR_INTEL1 = 0x756E6547; // "Genu"
@@ -27,238 +21,46 @@ private:
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
public:
// X86 cacheline size effectively has to be hardcoded to 64
// if we report anything differently then applications are likely to break
constexpr static uint64_t CACHELINE_SIZE = 64;
void Init(FEXCore::Context::Context *ctx);
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) {
if (Function < Primary.size()) {
const auto Handler = Primary[Function];
return (this->*Handler)(Leaf);
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, [[maybe_unused]] uint32_t Leaf) {
auto Handler = FunctionHandlers.find(Function);
if (Handler == FunctionHandlers.end()) {
#ifndef NDEBUG
LogMan::Msg::E("Unhandled CPU ID function, 0x%x", Function);
#endif
return Function_Reserved();
}
constexpr uint32_t HypervisorBase = 0x4000'0000;
if (Function >= HypervisorBase && Function < (HypervisorBase + Hypervisor.size())) {
const auto Handler = Hypervisor[Function - HypervisorBase];
return (this->*Handler)(Leaf);
}
constexpr uint32_t ExtendedBase = 0x8000'0000;
if (Function >= ExtendedBase && Function < (ExtendedBase + Extended.size())) {
const auto Handler = Extended[Function - ExtendedBase];
return (this->*Handler)(Leaf);
}
return Function_Reserved(Leaf);
return Handler->second();
}
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
if (Function == 0x8000'0002U)
return Function_8000_0002h(Leaf, CPU % PerCPUData.size());
else if (Function == 0x8000'0003U)
return Function_8000_0003h(Leaf, CPU % PerCPUData.size());
else
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
}
private:
FEXCore::Context::Context *CTX;
bool Hybrid{};
FEX_CONFIG_OPT(Cores, THREADS);
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
struct CPUData {
const char *ProductName{};
#ifdef _M_ARM_64
uint32_t MIDR{};
#endif
bool IsBig{};
};
std::vector<CPUData> PerCPUData{};
using FunctionHandler = std::function<FEXCore::CPUID::FunctionResults()>;
void RegisterFunction(uint32_t Function, FunctionHandler Handler) {
FunctionHandlers[Function] = Handler;
}
std::unordered_map<uint32_t, FunctionHandler> FunctionHandlers;
// Functions
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_01h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_02h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_04h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_06h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_07h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_4000_0001h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_0h();
FEXCore::CPUID::FunctionResults Function_01h();
FEXCore::CPUID::FunctionResults Function_02h();
FEXCore::CPUID::FunctionResults Function_06h();
FEXCore::CPUID::FunctionResults Function_07h();
FEXCore::CPUID::FunctionResults Function_15h();
FEXCore::CPUID::FunctionResults Function_8000_0000h();
FEXCore::CPUID::FunctionResults Function_8000_0001h();
FEXCore::CPUID::FunctionResults Function_8000_0002h();
FEXCore::CPUID::FunctionResults Function_8000_0003h();
FEXCore::CPUID::FunctionResults Function_8000_0004h();
FEXCore::CPUID::FunctionResults Function_8000_0005h();
FEXCore::CPUID::FunctionResults Function_8000_0006h();
FEXCore::CPUID::FunctionResults Function_8000_0007h();
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf, uint32_t CPU);
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf, uint32_t CPU);
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf, uint32_t CPU);
FEXCore::CPUID::FunctionResults Function_8000_0005h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0008h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
void SetupHostHybridFlag();
static constexpr std::array<FunctionHandler, 27> Primary = {
// 0: Highest function parameter and ID
&CPUIDEmu::Function_0h,
// 1: Processor info
&CPUIDEmu::Function_01h,
// 2: Cache and TLB info
&CPUIDEmu::Function_02h,
// 3: Serial Number(previously), now reserved
&CPUIDEmu::Function_Reserved,
#ifndef CPUID_AMD
// 4: Deterministic cache parameters for each level
&CPUIDEmu::Function_04h,
#else
&CPUIDEmu::Function_Reserved,
#endif
// 5: Monitor/mwait
&CPUIDEmu::Function_Reserved,
// 6: Thermal and power management
&CPUIDEmu::Function_06h,
// 7: Extended feature flags
&CPUIDEmu::Function_07h,
// 0x08: Reserved?
&CPUIDEmu::Function_Reserved,
// 9: Direct Cache Access information
&CPUIDEmu::Function_Reserved,
// 0x0A: Architectural performance monitoring
&CPUIDEmu::Function_Reserved,
// 0x0B: Extended topology enumeration
&CPUIDEmu::Function_Reserved,
// 0x0C: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x0D: Processor extended state enumeration
&CPUIDEmu::Function_0Dh,
// 0x0E: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x0F: Intel RDT monitoring
&CPUIDEmu::Function_Reserved,
// 0x10: Intel RDT allocation enumeration
&CPUIDEmu::Function_Reserved,
// 0x12: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x12: Intel SGX capability enumeration
&CPUIDEmu::Function_Reserved,
// 0x13: Reserved
&CPUIDEmu::Function_Reserved,
// 0x14: Intel Processor trace
&CPUIDEmu::Function_Reserved,
#ifndef CPUID_AMD
// Timestamp counter information
// Doesn't exist on AMD hardware
&CPUIDEmu::Function_15h,
#else
&CPUIDEmu::Function_Reserved,
#endif
// 0x16: Processor frequency information
&CPUIDEmu::Function_Reserved,
// 0x17: SoC vendor attribute enumeration
&CPUIDEmu::Function_Reserved,
// 0x18: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x19: Reserved?
&CPUIDEmu::Function_Reserved,
#ifndef CPUID_AMD
// 0x1A: Hybrid Information Sub-leaf
&CPUIDEmu::Function_1Ah,
#else
&CPUIDEmu::Function_Reserved,
#endif
};
static constexpr std::array<FunctionHandler, 2> Hypervisor = {
// Hypervisor CPUID information leaf
&CPUIDEmu::Function_4000_0000h,
// FEX-Emu specific leaf
&CPUIDEmu::Function_4000_0001h,
};
static constexpr std::array<FunctionHandler, 32> Extended = {
// Largest extended function number
&CPUIDEmu::Function_8000_0000h,
// Processor vendor
&CPUIDEmu::Function_8000_0001h,
// Processor brand string
&CPUIDEmu::Function_8000_0002h,
// Processor brand string continued
&CPUIDEmu::Function_8000_0003h,
// Processor brand string continued
&CPUIDEmu::Function_8000_0004h,
#ifdef CPUID_AMD
// 0x8000'0005: L1 Cache and TLB identifiers
&CPUIDEmu::Function_8000_0005h,
#else
&CPUIDEmu::Function_Reserved,
#endif
// 0x8000'0006: L2 Cache identifiers
&CPUIDEmu::Function_8000_0006h,
// 0x8000'0007: Advanced power management information
&CPUIDEmu::Function_8000_0007h,
// 0x8000'0008: Virtual and physical address sizes
&CPUIDEmu::Function_8000_0008h,
// 0x8000'0009: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'000A: SVM Revision
&CPUIDEmu::Function_Reserved,
// 0x8000'000B: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'000C: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'000D: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'000E: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'000F: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0010: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0011: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0012: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0013: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0014: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0015: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0016: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0017: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0018: Reserved?
&CPUIDEmu::Function_Reserved,
// 0x8000'0019: TLB 1GB page identifiers
&CPUIDEmu::Function_8000_0019h,
// 0x8000'001A: Performance optimization identifiers
&CPUIDEmu::Function_Reserved,
// 0x8000'001B: Instruction based sampling identifiers
&CPUIDEmu::Function_Reserved,
// 0x8000'001C: Lightweight profiling capabilities
&CPUIDEmu::Function_Reserved,
#ifdef CPUID_AMD
// 0x8000'001D: Cache properties
&CPUIDEmu::Function_8000_001Dh,
#else
&CPUIDEmu::Function_Reserved,
#endif
// 0x8000'001E: Extended APIC ID
&CPUIDEmu::Function_Reserved,
// 0x8000'001F: AMD Secure Encryption
&CPUIDEmu::Function_Reserved,
};
FEXCore::CPUID::FunctionResults Function_Reserved();
};
}
@@ -0,0 +1,168 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/CompileService.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/OpcodeDispatcher.h"
namespace FEXCore {
static void* ThreadHandler(void *Arg) {
FEXCore::CompileService *This = reinterpret_cast<FEXCore::CompileService*>(Arg);
This->ExecutionThread();
return nullptr;
}
CompileService::CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
: CTX {ctx}
, ParentThread {Thread} {
CompileThreadData = std::make_unique<FEXCore::Core::InternalThreadState>();
CompileThreadData->IsCompileService = true;
// We need a compiler for this work thread
CTX->InitializeCompiler(CompileThreadData.get(), true);
CompileThreadData->CPUBackend->CopyNecessaryDataForCompileThread(ParentThread->CPUBackend.get());
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
}
void CompileService::Initialize() {
// Share CompileService which = this
CompileThreadData->CompileService = ParentThread->CompileService;
}
void CompileService::Shutdown() {
ShuttingDown = true;
// Kick the working thread
StartWork.NotifyAll();
WorkerThread->join(nullptr);
}
void CompileService::ClearCache(FEXCore::Core::InternalThreadState *Thread) {
// On cache clear we need to spin down the execution thread to ensure it isn't trying to give us more work items
if (CompileMutex.try_lock()) {
// We can only clear these things if we pulled the compile mutex
// Grab the work queue and clear it
// We don't need to grab the queue mutex since this thread will no longer receive any work events
// Threads are bounded 1:1
while (WorkQueue.size()) {
WorkItem *Item = WorkQueue.front();
WorkQueue.pop();
delete Item;
}
// Go through the garbage collection array and clear it
// It's safe to clear things that aren't marked safe since we are clearing cache
if (GCArray.size()) {
// Clean up our GC array
for (auto it = GCArray.begin(); it != GCArray.end();) {
delete *it;
it = GCArray.erase(it);
}
}
LogMan::Throw::A(CompileThreadData->LocalIRCache.size() == 0, "Compile service must never have LocalIRCache");
CompileMutex.unlock();
}
// Clear the inverse cache of what is calling us from the Context ClearCache routine
auto SelectedThread = Thread->IsCompileService ? ParentThread : Thread;
SelectedThread->LookupCache->ClearCache();
SelectedThread->CPUBackend->ClearCache();
}
CompileService::WorkItem *CompileService::CompileCode(uint64_t RIP) {
// Tell the worker thread to compile code for us
WorkItem *Item = new WorkItem{};
Item->RIP = RIP;
{
// Fill the threads work queue
std::scoped_lock<std::mutex> lk(QueueMutex);
WorkQueue.emplace(Item);
}
// Notify the thread that it has more work
StartWork.NotifyAll();
return Item;
}
void CompileService::ExecutionThread() {
// Ignore signals coming from the guest
CTX->SignalDelegation->MaskThreadSignals();
// Set our thread name so we can see its relation
char ThreadName[16]{};
snprintf(ThreadName, 16, "%ld-CS", ParentThread->ThreadManager.TID.load());
pthread_setname_np(pthread_self(), ThreadName);
while (true) {
// Wait for work
StartWork.Wait();
if (ShuttingDown.load()) {
break;
}
std::scoped_lock<std::mutex> lk(CompileMutex);
size_t WorkItems{};
do {
// Grab a work item
WorkItem *Item{};
{
std::scoped_lock<std::mutex> lk(QueueMutex);
WorkItems = WorkQueue.size();
if (WorkItems) {
Item = WorkQueue.front();
WorkQueue.pop();
}
}
// If we had a work item then work on it
if (Item) {
// Make sure it's not in lookup cache by accident
LogMan::Throw::A(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
// Code isn't in cache, compile now
// Set our thread state's RIP
CompileThreadData->CurrentFrame->State.rip = Item->RIP;
auto [CodePtr, IRList, DebugData, RAData, Generated, StartAddr, Length] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
LogMan::Throw::A(Generated == true, "Compile Service doesn't have IR Cache");
if (!CodePtr) {
// XXX: We currently have the expectation that compile service code will be significantly smaller than regular thread's code
ERROR_AND_DIE("Couldn't compile code for thread at RIP: 0x%lx", Item->RIP);
}
Item->CodePtr = CodePtr;
Item->IRList = IRList;
Item->DebugData = DebugData;
Item->RAData = RAData;
Item->StartAddr = StartAddr;
Item->Length = Length;
GCArray.emplace_back(Item);
Item->ServiceWorkDone.NotifyAll();
}
} while (WorkItems != 0);
if (GCArray.size()) {
// Clean up our GC array
for (auto it = GCArray.begin(); it != GCArray.end();) {
if ((*it)->SafeToClear) {
delete *it;
it = GCArray.erase(it);
}
else {
++it;
}
}
}
}
}
}
+66
View File
@@ -0,0 +1,66 @@
#pragma once
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/Threads.h>
#include <memory>
#include <thread>
#include <unordered_map>
#include <queue>
namespace FEXCore {
namespace Context {
struct Context;
}
namespace Core {
struct InternalThreadState;
}
namespace IR {
class RegisterAllocationData;
};
class CompileService final {
public:
CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
void Initialize();
void Shutdown();
struct WorkItem {
// Incoming
uint64_t RIP{};
// Outgoing
void *CodePtr{};
FEXCore::IR::IRListView *IRList{};
FEXCore::IR::RegisterAllocationData *RAData{};
FEXCore::Core::DebugData *DebugData{};
uint64_t StartAddr;
uint64_t Length;
// Communication
Event ServiceWorkDone{};
std::atomic_bool SafeToClear{};
};
WorkItem *CompileCode(uint64_t RIP);
void ClearCache(FEXCore::Core::InternalThreadState *Thread);
// Public for threading
void ExecutionThread();
private:
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *ParentThread;
std::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
std::unique_ptr<FEXCore::Core::InternalThreadState> CompileThreadData;
std::mutex QueueMutex{};
std::mutex CompileMutex{};
std::queue<WorkItem*> WorkQueue{};
std::vector<WorkItem*> GCArray{};
Event StartWork{};
std::atomic_bool ShuttingDown{false};
};
}
File diff suppressed because it is too large. Load diff
@@ -1,60 +1,31 @@
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/X86HelperGen.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <array>
#include <bit>
#include <cmath>
#include <cstddef>
#include <cstdint>
#include <memory>
#include <aarch64/assembler-aarch64.h>
#include <aarch64/constants-aarch64.h>
#include <aarch64/cpu-aarch64.h>
#include <aarch64/operands-aarch64.h>
#include <code-buffer-vixl.h>
#include <platform-vixl.h>
#include <sys/syscall.h>
#include <unistd.h>
#include "aarch64/assembler-aarch64.h"
#include "aarch64/cpu-aarch64.h"
#include "aarch64/disasm-aarch64.h"
namespace FEXCore::CPU {
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
using namespace vixl;
using namespace vixl::aarch64;
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
#ifdef VIXL_SIMULATOR
, Simulator {&Decoder}
#endif
{
#ifdef VIXL_SIMULATOR
// Hardcode a 256-bit vector width if we are running in the simulator.
Simulator.SetVectorLengthInBits(256);
#endif
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
#define STATE x28
EmitDispatcher();
}
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
: Dispatcher(ctx, Thread), Arm64Emitter(MAX_DISPATCHER_CODE_SIZE) {
SRAEnabled = config.StaticRegisterAssignment;
SetAllowAssembler(true);
void Arm64Dispatcher::EmitDispatcher() {
#ifdef VIXL_DISASSEMBLER
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
#endif
DispatchPtr = GetCursorAddress<AsmDispatch>();
auto Buffer = GetBuffer();
DispatchPtr = Buffer->GetOffsetAddress<CPUBackend::AsmDispatch>(GetCursorOffset());
// while (true) {
// Ptr = FindBlock(RIP)
@@ -64,9 +35,15 @@ void Arm64Dispatcher::EmitDispatcher() {
// Ptr();
// }
ARMEmitter::ForwardLabel l_CTX;
ARMEmitter::ForwardLabel l_Sleep;
ARMEmitter::ForwardLabel l_CompileBlock;
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
Literal l_VirtualMemory {VirtualMemorySize};
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
Literal l_L1Ptr {Thread->LookupCache->GetL1Pointer()};
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
Literal l_CompileBlock {GetCompileBlockPtr()};
Literal l_ExitFunctionLink {config.ExitFunctionLink};
Literal l_ExitFunctionLinkThis {config.ExitFunctionLinkThis};
// Push all the register we need to save
PushCalleeSavedRegisters();
@@ -74,119 +51,136 @@ void Arm64Dispatcher::EmitDispatcher() {
// Push our memory base to the correct register
// Move our thread pointer to the correct register
// This is passed in to parameter 0 (x0)
mov(STATE, ARMEmitter::XReg::x0);
mov(STATE, x0);
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
// regardless of where we were in the stack
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::rsp, 0);
str(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
add(x0, sp, 0);
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
AbsoluteLoopTopAddressFillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (config.StaticRegisterAllocation) {
if (SRAEnabled) {
FillStaticRegs();
}
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
ARMEmitter::BiDirectionalLabel FullLookup{};
ARMEmitter::BiDirectionalLabel CallBlock{};
ARMEmitter::BackwardLabel LoopTop{};
aarch64::Label FullLookup{};
aarch64::Label LoopTop{};
aarch64::Label ExitSpillSRA{};
aarch64::Label ThreadPauseHandler{};
Bind(&LoopTop);
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
bind(&LoopTop);
AbsoluteLoopTopAddress = GetLabelAddress<uint64_t>(&LoopTop);
// Load in our RIP
// Don't modify x2 since it contains our RIP once the block doesn't exist
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
auto RipReg = x2;
auto RipReg = ARMEmitter::XReg::x2;
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
if (!config.ExecuteBlocksWithCall) {
// L1 Cache
ldr(x0, &l_L1Ptr);
// L1 Cache
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, RipReg.R());
b(ARMEmitter::Condition::CC_NE, &FullLookup);
br(ARMEmitter::Reg::r3);
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x3, Shift::LSL, 4));
ldp(x1, x0, MemOperand(x0));
cmp(x0, RipReg);
b(&FullLookup, Condition::ne);
br(x1);
}
// L1C check failed, do a full lookup
Bind(&FullLookup);
bind(&FullLookup);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
ldr(x0, &l_PagePtr);
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), VirtualMemorySize - 1);
if (__builtin_popcountl(VirtualMemorySize) == 1) {
and_(x3, RipReg, Thread->LookupCache->GetVirtualMemorySize() - 1);
}
else {
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
ldr(x3, &l_VirtualMemory);
and_(x3, RipReg, x3);
}
#ifdef VIXL_SIMULATOR
// VIXL simulator can't run syscalls.
constexpr bool SignalSafeCompile = false;
#else
constexpr bool SignalSafeCompile = true;
#endif
ARMEmitter::ForwardLabel NoBlock;
aarch64::Label NoBlock;
{
// Offset the address and add to our page pointer
lsr(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 12);
lsr(x1, x3, 12);
// Load the pointer from the offset
ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ExtendedType::LSL_64, 3);
ldr(x0, MemOperand(x0, x1, Shift::LSL, 3));
// If page pointer is zero then we have no block
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &NoBlock);
cbz(x0, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 0x0FFF);
and_(x1, x3, 0x0FFF);
// Shift the offset by the size of the block cache entry
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
add(x0, x0, Operand(x1, Shift::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry))));
// The the full LookupCacheEntry with a single LDP.
// Check the guest address first to ensure it maps to the address we are currently at.
// Load the guest address first to ensure it maps to the address we are currently at
// This fixes aliasing problems
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
ldr(x1, MemOperand(x0, offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)));
cmp(x1, RipReg);
b(&NoBlock, Condition::ne);
// If the guest address doesn't match, Compile the block.
cmp(ARMEmitter::XReg::x1, RipReg);
b(ARMEmitter::Condition::CC_NE, &NoBlock);
// Check the host address to see if it matches, else compile the block.
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
// Now load the actual host block to execute if we can
ldr(x3, MemOperand(x0, offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)));
cbz(x3, &NoBlock);
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, 4);
stp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x2, ARMEmitter::Reg::r0);
// Jump to the block
br(ARMEmitter::Reg::r3);
if (!config.ExecuteBlocksWithCall) {
// update L1 cache
ldr(x0, &l_L1Ptr);
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x1, Shift::LSL, 4));
stp(x3, x2, MemOperand(x0));
br(x3);
} else {
mov(x0, STATE);
blr(x3);
}
}
if (config.ExecuteBlocksWithCall) {
// Interpreter continues execution here
if (CTX->GetGdbServerStatus()) {
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
ldr(x0, &l_CTX);
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
// If the value == 0 then branch to the top
cbz(x0, &LoopTop);
// Else we need to pause now
b(&ThreadPauseHandler);
}
else {
// Unconditionally loop to the top
// We will only stop on error when compiling a block or signal
b(&LoopTop);
}
}
}
{
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
bind(&ExitSpillSRA);
ThreadStopHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled)
SpillStaticRegs();
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
ThreadStopHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
PopCalleeSavedRegisters();
@@ -196,124 +190,44 @@ void Arm64Dispatcher::EmitDispatcher() {
}
{
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
ExitFunctionLinkerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled)
SpillStaticRegs();
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// Args:
// X0: SETMASK
// X1: Pointer to mask value (uint64_t)
// X2: Pointer to old mask value (uint64_t)
// X3: Size of mask, sizeof(uint64_t)
// X8: Syscall
ldr(x0, &l_ExitFunctionLinkThis);
mov(x1, STATE);
mov(x2, lr);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::Reg::rsp, -16);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
svc(0);
}
ldr(x3, &l_ExitFunctionLink);
blr(x3);
mov(ARMEmitter::XReg::x0, STATE);
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uintptr_t, void *, void *>(ARMEmitter::Reg::r2);
#else
blr(ARMEmitter::Reg::r2);
#endif
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
mov(ARMEmitter::XReg::x4, ARMEmitter::XReg::x0);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
svc(0);
// Bring stack back
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
mov(ARMEmitter::XReg::x0, ARMEmitter::XReg::x4);
}
if (config.StaticRegisterAllocation)
if (SRAEnabled)
FillStaticRegs();
br(ARMEmitter::Reg::r0);
br(x0);
}
// Need to create the block
{
Bind(&NoBlock);
bind(&NoBlock);
if (config.StaticRegisterAllocation)
ldr(x0, &l_CTX);
mov(x1, STATE);
ldr(x3, &l_CompileBlock);
if (SRAEnabled)
SpillStaticRegs();
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// Args:
// X0: SETMASK
// X1: Pointer to mask value (uint64_t)
// X2: Pointer to old mask value (uint64_t)
// X3: Size of mask, sizeof(uint64_t)
// X8: Syscall
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, -16);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
svc(0);
// Reload x2 to bring back RIP
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, 8);
}
ldr(ARMEmitter::XReg::x0, &l_CTX);
mov(ARMEmitter::XReg::x1, STATE);
ldr(ARMEmitter::XReg::x3, &l_CompileBlock);
// X2 contains our guest RIP
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<void, void *, uint64_t, void *>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
#endif
blr(x3); // { CTX, Frame, RIP}
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
svc(0);
// Bring stack back
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
}
if (config.StaticRegisterAllocation)
if (SRAEnabled)
FillStaticRegs();
b(&LoopTop);
}
{
SignalHandlerReturnAddress = GetCursorAddress<uint64_t>();
SignalHandlerReturnAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
// Now to get back to our old location we need to do a fault dance
// We can't use SIGTRAP here since gdb catches it and never gives it to the application!
@@ -321,71 +235,22 @@ void Arm64Dispatcher::EmitDispatcher() {
}
{
SignalHandlerReturnAddressRT = GetCursorAddress<uint64_t>();
// Now to get back to our old location we need to do a fault dance
// We can't use SIGTRAP here since gdb catches it and never gives it to the application!
hlt(0);
}
{
// Guest SIGILL handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
ThreadPauseHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled)
SpillStaticRegs();
hlt(0);
}
{
// Guest SIGTRAP handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
SpillStaticRegs();
brk(0);
}
{
// Guest Overflow handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
SpillStaticRegs();
// hlt/udf = SIGILL
// brk = SIGTRAP
// ??? = SIGSEGV
// Force a SIGSEGV by loading zero
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
}
{
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
if (config.StaticRegisterAllocation)
SpillStaticRegs();
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
bind(&ThreadPauseHandler);
ThreadPauseHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
// We are pausing, this means the frontend should be waiting for this thread to idle
// We will have faulted and jumped to this location at this point
// Call our sleep handler
ldr(ARMEmitter::XReg::x0, &l_CTX);
mov(ARMEmitter::XReg::x1, STATE);
ldr(ARMEmitter::XReg::x2, &l_Sleep);
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<void, void *, void *>(ARMEmitter::Reg::r2);
#else
blr(ARMEmitter::Reg::r2);
#endif
ldr(x0, &l_CTX);
mov(x1, STATE);
ldr(x2, &l_Sleep);
blr(x2);
PauseReturnInstruction = GetCursorAddress<uint64_t>();
PauseReturnInstruction = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
// Fault to start running again
hlt(0);
}
@@ -406,274 +271,93 @@ void Arm64Dispatcher::EmitDispatcher() {
// On return to the thunk, the thunk can get whatever its return value is from the thread context depending on ABI handling on its end
// When the thunk itself returns, it'll do its regular return logic there
// void ReentrantCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
CallbackPtr = GetCursorAddress<JITCallback>();
CallbackPtr = Buffer->GetOffsetAddress<CPUBackend::JITCallback>(GetCursorOffset());
// We expect the thunk to have previously pushed the registers it was using
PushCalleeSavedRegisters();
// First thing we need to move the thread state pointer back in to our register
mov(STATE, ARMEmitter::XReg::x0);
mov(STATE, x0);
// Make sure to adjust the refcounter so we don't clear the cache now
ldr(ARMEmitter::WReg::w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
add(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, 1);
str(ARMEmitter::WReg::w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
LoadConstant(x0, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
ldr(w2, MemOperand(x0));
add(w2, w2, 1);
str(w2, MemOperand(x0));
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->X86CodeGen.CallbackReturn);
LoadConstant(x0, CTX->X86CodeGen.CallbackReturn);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, 16);
str(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
sub(x2, x2, 16);
str(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
// Store the trampoline to the guest stack
// Guest stack is now correctly misaligned after a regular call instruction
str(ARMEmitter::XReg::x0, ARMEmitter::Reg::r2, 0);
str(x0, MemOperand(x2));
// Store RIP to the context state
str(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, State.rip));
str(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
// load static regs
if (config.StaticRegisterAllocation)
if (SRAEnabled)
FillStaticRegs();
// Now go back to the regular dispatcher loop
b(&LoopTop);
}
{
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
place(&l_VirtualMemory);
place(&l_PagePtr);
place(&l_L1Ptr);
place(&l_CTX);
place(&l_Sleep);
place(&l_CompileBlock);
place(&l_ExitFunctionLink);
place(&l_ExitFunctionLinkThis);
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs();
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r3);
#endif
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs();
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r3);
#endif
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs();
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r3);
#endif
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs();
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r3);
#endif
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
Bind(&l_CTX);
dc64(reinterpret_cast<uintptr_t>(CTX));
Bind(&l_Sleep);
dc64(reinterpret_cast<uint64_t>(SleepThread));
Bind(&l_CompileBlock);
dc64(GetCompileBlockPtr());
FinalizeCode();
Start = reinterpret_cast<uint64_t>(DispatchPtr);
End = GetCursorAddress<uint64_t>();
ClearICache(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
End = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
GetBuffer()->SetExecutable();
if (CTX->Config.BlockJITNaming()) {
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
}
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
}
#ifdef VIXL_DISASSEMBLER
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
#if ENABLE_JITSYMBOLS
std::string Name = "Dispatch_" + std::to_string(::gettid());
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
#endif
}
#ifdef VIXL_SIMULATOR
void Arm64Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(DispatchPtr));
void Arm64Dispatcher::SpillSRA(void *ucontext) {
for(int i = 0; i < SRA64.size(); i++) {
ThreadState->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
}
// TODO: Also recover FPRs, not sure where the neon context is
// This is usually not needed
/*
for(int i = 0; i < SRAFPR.size(); i++) {
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
}
*/
}
void Arm64Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.WriteXRegister(1, RIP);
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(CallbackPtr));
#ifdef _M_ARM_64
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
DispatcherConfig config;
config.ExecuteBlocksWithCall = true;
Dispatcher = new Arm64Dispatcher(ctx, Thread, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
// TODO: It feels wrong to initialize this way
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
}
#endif
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
FEXCore::ARMEmitter::Emitter emit{CodeBuffer, MaxGDBPauseCheckSize};
ARMEmitter::ForwardLabel RunBlock;
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(FEXCore::Context::Context::Config.RunningMode) == 4, "This is expected to be size of 4");
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Thread));
emit.ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, offsetof(FEXCore::Core::InternalThreadState, CTX)); // Get Context
emit.ldr(ARMEmitter::WReg::w0, ARMEmitter::Reg::r0, offsetof(FEXCore::Context::Context, Config.RunningMode));
// If the value == 0 then we don't need to stop
emit.cbz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, &RunBlock);
{
ARMEmitter::ForwardLabel l_GuestRIP;
// Make sure RIP is syncronized to the context
emit.ldr(ARMEmitter::XReg::x0, &l_GuestRIP);
emit.str(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, State.rip));
// Stop the thread
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
emit.br(ARMEmitter::Reg::r0);
emit.Bind(&l_GuestRIP);
emit.dc64(GuestRIP);
}
emit.Bind(&RunBlock);
auto UsedBytes = emit.GetCursorOffset();
emit.ClearICache(CodeBuffer, UsedBytes);
return UsedBytes;
}
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
FEXCore::ARMEmitter::Emitter emit{CodeBuffer, MaxInterpreterTrampolineSize};
ARMEmitter::ForwardLabel InlineIRData;
emit.mov(ARMEmitter::XReg::x0, STATE);
emit.adr(ARMEmitter::Reg::r1, &InlineIRData);
emit.ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
emit.blr(ARMEmitter::Reg::r3);
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
emit.br(ARMEmitter::Reg::r0);
emit.Bind(&InlineIRData);
auto UsedBytes = emit.GetCursorOffset();
emit.ClearICache(CodeBuffer, UsedBytes);
return UsedBytes;
}
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
for (size_t i = 0; i < SRA64.size(); i++) {
if (IgnoreMask & (1U << SRA64[i].Idx())) {
// Skip this one, it's already spilled
continue;
}
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].Idx());
}
if (EmitterCTX->HostFeatures.SupportsAVX) {
for (size_t i = 0; i < SRAFPR.size(); i++) {
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].Idx());
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &FPR, sizeof(__uint128_t));
}
} else {
for (size_t i = 0; i < SRAFPR.size(); i++) {
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].Idx());
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &FPR, sizeof(__uint128_t));
}
}
}
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto &Common = Thread->CurrentFrame->Pointers.Common;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
Common.SignalReturnHandler = SignalHandlerReturnAddress;
Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
AArch64.LUDIVHandler = LUDIVHandlerAddress;
AArch64.LDIVHandler = LDIVHandlerAddress;
AArch64.LUREMHandler = LUREMHandlerAddress;
AArch64.LREMHandler = LREMHandlerAddress;
}
}
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
return std::make_unique<Arm64Dispatcher>(CTX, Config);
}
}
@@ -3,51 +3,16 @@
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#endif
namespace FEXCore::Context {
struct Context;
}
namespace FEXCore::Core {
struct InternalThreadState;
}
#define STATE_PTR(STATE_TYPE, FIELD) \
STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
#include "aarch64/assembler-aarch64.h"
namespace FEXCore::CPU {
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
public:
Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
#ifdef VIXL_SIMULATOR
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) override;
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) override;
#endif
void EmitDispatcher();
Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
protected:
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
private:
// Long division helpers
uint64_t LUDIVHandlerAddress{};
uint64_t LDIVHandlerAddress{};
uint64_t LUREMHandlerAddress{};
uint64_t LREMHandlerAddress{};
#ifdef VIXL_SIMULATOR
vixl::aarch64::Decoder Decoder;
vixl::aarch64::Simulator Simulator;
#endif
void SpillSRA(void *ucontext) override;
};
}
}
File diff suppressed because it is too large. Load diff
+35 -128
View File
@@ -1,37 +1,26 @@
#pragma once
#include <FEXCore/Core/CPUBackend.h>
#include "Interface/Core/ArchHelpers/MContext.h"
#include <FEXCore/Core/SignalDelegator.h>
#include "Interface/Context/Context.h"
#include <cstdint>
#include <signal.h>
#include <stddef.h>
#include <stack>
#include <tuple>
#include <vector>
namespace FEXCore {
struct GuestSigAction;
}
namespace FEXCore::Core {
struct CpuStateFrame;
struct InternalThreadState;
}
namespace FEXCore::Context {
struct Context;
}
namespace FEXCore::CPU {
struct DispatcherConfig {
bool StaticRegisterAllocation = false;
bool ExecuteBlocksWithCall = false;
uintptr_t ExitFunctionLink = 0;
uintptr_t ExitFunctionLinkThis = 0;
bool StaticRegisterAssignment = false;
};
class Dispatcher {
public:
virtual ~Dispatcher() = default;
CPUBackend::AsmDispatch DispatchPtr;
CPUBackend::JITCallback CallbackPtr;
FEXCore::Context::Context::IntCallbackReturn ReturnPtr;
/**
* @name Dispatch Helper functions
@@ -44,134 +33,52 @@ public:
uint64_t ThreadPauseHandlerAddressSpillSRA{};
uint64_t ExitFunctionLinkerAddress{};
uint64_t SignalHandlerReturnAddress{};
uint64_t SignalHandlerReturnAddressRT{};
uint64_t GuestSignal_SIGILL{};
uint64_t GuestSignal_SIGTRAP{};
uint64_t GuestSignal_SIGSEGV{};
uint64_t IntCallbackReturnAddress{};
uint64_t PauseReturnInstruction{};
/** @} */
uint32_t SignalHandlerRefCounter{};
uint64_t Start{};
uint64_t End{};
bool HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
bool HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
bool HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
bool HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
bool HandleSIGILL(int Signal, void *info, void *ucontext);
bool HandleSignalPause(int Signal, void *info, void *ucontext);
bool IsAddressInDispatcher(uint64_t Address) const {
void RegisterCodeBuffer(uint8_t* start, size_t size) {
CodeBuffers.emplace_back(reinterpret_cast<uint64_t>(start),
reinterpret_cast<uint64_t>(start + size));
}
void RemoveCodeBuffer(uint8_t* start);
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true);
bool IsAddressInDispatcher(uint64_t Address) {
return Address >= Start && Address < End;
}
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
// These are across all arches for now
static constexpr size_t MaxGDBPauseCheckSize = 128;
static constexpr size_t MaxInterpreterTrampolineSize = 128;
virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0;
virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0;
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
virtual void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
DispatchPtr(Frame);
}
virtual void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
CallbackPtr(Frame, RIP);
}
protected:
Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &Config)
Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
: CTX {ctx}
, config {Config}
{}
, ThreadState {Thread} {}
void RestoreFrame_x64(ArchHelpers::Context::ContextBackup* Context, FEXCore::Core::CpuStateFrame *Frame, void *ucontext);
void RestoreFrame_ia32(ArchHelpers::Context::ContextBackup* Context, FEXCore::Core::CpuStateFrame *Frame, void *ucontext);
void RestoreRTFrame_ia32(ArchHelpers::Context::ContextBackup* Context, FEXCore::Core::CpuStateFrame *Frame, void *ucontext);
void StoreThreadState(int Signal, void *ucontext);
void RestoreThreadState(void *ucontext);
std::stack<uint64_t> SignalFrames;
const bool incomplete_guest_restorer_support = false;
///< Setup the signal frame for x64.
uint64_t SetupFrame_x64(FEXCore::Core::InternalThreadState *Thread, ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame *Frame,
int Signal, siginfo_t *HostSigInfo, void *ucontext,
GuestSigAction *GuestAction, stack_t *GuestStack,
uint64_t NewGuestSP, const uint32_t eflags);
///< Setup the signal frame for a 32-bit signal without SA_SIGINFO.
uint64_t SetupFrame_ia32(ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame *Frame,
int Signal, siginfo_t *HostSigInfo, void *ucontext,
GuestSigAction *GuestAction, stack_t *GuestStack,
uint64_t NewGuestSP, const uint32_t eflags);
///< Setup the signal frame for a 32-bit signal with SA_SIGINFO.
uint64_t SetupRTFrame_ia32(ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame *Frame,
int Signal, siginfo_t *HostSigInfo, void *ucontext,
GuestSigAction *GuestAction, stack_t *GuestStack,
uint64_t NewGuestSP, const uint32_t eflags);
ArchHelpers::Context::ContextBackup* StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext);
enum class RestoreType {
TYPE_REALTIME, ///< Signal restore type is from a `realtime` signal.
TYPE_NONREALTIME, ///< Signal restore type is from a `non-realtime` signal.
TYPE_PAUSE, ///< Signal restore type is from a GDB pause event.
};
/*
* Signal frames on 32-bit architecture needs to match exactly how the kernel generates the frame.
* This is because large parts of the signal frame definition is part of the UAPI.
* This means that when FEX sets up the signal frame, it needs to match the UAPI stack setup.
*
* The two signal stack frame types below describe the two different 32-bit frame types.
*/
// The 32-bit non-realtime signal frame.
// This frame type is used when the guest signal is used without the `SA_SIGINFO` flag.
struct SigFrame_i32 {
uint32_t pretcode; ///< sigreturn return branch point.
int32_t Signal; ///< The signal hit.
FEXCore::x86::sigcontext sc; ///< The signal context.
x86::_libc_fpstate fpstate_unused; ///< Unused fpstate. Retained for backwards compatibility.
uint32_t extramask[1]; ///< Upper 32-bits of the signal mask. Lower 32-bits is in the sigcontext.
char retcode[8]; ///< Unused but needs to be filled. GDB seemingly uses as a debug marker.
///< FP state now follows after this.
};
// The 32-bit realtime signal frame.
// This frame type is used when the guest signal is used with the `SA_SIGINFO` flag.
struct RTSigFrame_i32 {
uint32_t pretcode; ///< sigreturn return branch point.
int32_t Signal; ///< The signal hit.
uint32_t pinfo; ///< Pointer to siginfo_t
uint32_t puc; ///< Pointer to ucontext_t
FEXCore::x86::siginfo_t info;
FEXCore::x86::ucontext_t uc;
char retcode[8]; ///< Unused but needs to be filled. GDB seemingly uses as a debug marker.
///< FP state now follows after this.
};
void RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext, RestoreType Type);
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
virtual void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {}
bool SRAEnabled = false;
virtual void SpillSRA(void *ucontext) {}
FEXCore::Context::Context *CTX;
DispatcherConfig config;
FEXCore::Core::InternalThreadState *ThreadState;
static void SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuStateFrame *Frame);
static uint64_t GetCompileBlockPtr();
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
AsmDispatch DispatchPtr;
JITCallback CallbackPtr;
private:
std::vector<std::tuple<uint64_t, uint64_t>> CodeBuffers; // Start, End
};
}
}
@@ -1,43 +1,22 @@
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Context/Context.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <cmath>
#include <memory>
#include <stddef.h>
#include <stdint.h>
#include <sys/mman.h>
#include <xbyak/xbyak.h>
#define STATE_PTR(STATE_TYPE, FIELD) \
[STATE + offsetof(FEXCore::Core::STATE_TYPE, FIELD)]
namespace FEXCore::CPU {
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
#define STATE r14
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
: Dispatcher(ctx, config)
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
nullptr) {
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "X86 dispatcher does not support SRA");
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
: Dispatcher(ctx, Thread)
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE) {
using namespace Xbyak;
using namespace Xbyak::util;
DispatchPtr = getCurr<AsmDispatch>();
DispatchPtr = getCurr<CPUBackend::AsmDispatch>();
// Temp registers
// rax, rcx, rdx, rsi, r8, r9,
@@ -83,7 +62,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
// regardless of where we were in the stack
mov(qword STATE_PTR(CpuStateFrame, ReturningStackLocation), rsp);
mov(qword [rdi + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)], rsp);
Label LoopTop;
Label FullLookup;
@@ -96,26 +75,27 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
{
// Load our RIP
mov(rdx, qword STATE_PTR(CPUState, rip));
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
// L1 Cache
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
mov(rax, rdx);
if (!config.ExecuteBlocksWithCall)
{
// L1 Cache
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(rax, rdx);
and_(rax, LookupCache::L1_ENTRIES_MASK);
shl(rax, 4);
cmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)], rdx);
jne(FullLookup);
jmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)]);
and_(rax, LookupCache::L1_ENTRIES_MASK);
shl(rax, 4);
cmp(qword[r13 + rax + 8], rdx);
jne(FullLookup);
jmp(qword[r13 + rax + 0]);
}
L(FullLookup);
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
mov(r13, Thread->LookupCache->GetPagePointer());
// Full lookup
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
mov(rax, rdx);
mov(rbx, VirtualMemorySize - 1);
mov(rbx, Thread->LookupCache->GetVirtualMemorySize() - 1);
and_(rax, rbx);
shr(rax, 12);
@@ -142,15 +122,39 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
je(NoBlock);
// Update L1
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
mov(rcx, rdx);
and_(rcx, LookupCache::L1_ENTRIES_MASK);
shl(rcx, 1);
mov(qword[r13 + rcx*8 + 8], rdx);
mov(qword[r13 + rcx*8 + 0], rax);
if (config.ExecuteBlocksWithCall) {
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(rcx, rdx);
and_(rcx, LookupCache::L1_ENTRIES_MASK);
shl(rcx, 1);
mov(qword[r13 + rcx*8 + 8], rdx);
mov(qword[r13 + rcx*8 + 0], rax);
}
// Real block if we made it here
jmp(rax);
if (!config.ExecuteBlocksWithCall) {
jmp(rax);
} else {
mov(rdi, STATE);
call(rax);
if (CTX->GetGdbServerStatus()) {
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
mov(rax, qword [STATE + (offsetof(FEXCore::Core::InternalThreadState, CTX))]);
// If the value == 0 then branch to the top
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
je(LoopTop);
// Else we need to pause now
jmp(ThreadPauseHandler);
ud2();
}
else {
jmp(LoopTop);
}
}
}
{
@@ -169,122 +173,40 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
ret();
}
constexpr bool SignalSafeCompile = true;
// Block creation
{
L(NoBlock);
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// RDI: SETMASK
// RSI: Pointer to mask value (uint64_t)
// RDX: Pointer to old mask value (uint64_t)
// R10: Size of mask, sizeof(uint64_t)
// RAX: Syscall
using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
// Backup rdx
mov(r9, rdx);
mov(rdi, ~0ULL);
sub(rsp, 16);
mov(qword [rsp], rdi);
mov(qword [rsp + 8], rdi);
mov(rdi, SIG_SETMASK);
mov(rsi, rsp);
mov(rdx, rsp);
mov(r10, 8);
mov(rax, SYS_rt_sigprocmask);
syscall();
mov(rdx, r9);
}
PtrCast Ptr;
Ptr.ClassPtr = &FEXCore::Context::Context::CompileBlock;
// {rdi, rsi, rdx}
mov(rdi, reinterpret_cast<uint64_t>(CTX));
mov(rsi, STATE);
mov(rax, GetCompileBlockPtr());
mov(rax, Ptr.Data);
call(rax);
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
// Backup rdx
mov(r9, rdx);
mov(rdi, SIG_SETMASK);
mov(rsi, rsp);
mov(rdx, 0); // Don't care about result
mov(r10, 8);
mov(rax, SYS_rt_sigprocmask);
syscall();
// Bring stack back
add(rsp, 16);
mov(rdx, r9);
}
// rdx already contains RIP here
jmp(LoopTop);
}
{
ExitFunctionLinkerAddress = getCurr<uint64_t>();
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// RDI: SETMASK
// RSI: Pointer to mask value (uint64_t)
// RDX: Pointer to old mask value (uint64_t)
// R10: Size of mask, sizeof(uint64_t)
// RAX: Syscall
// {rdi, rsi, rdx}
mov(rdi, config.ExitFunctionLinkThis);
mov(rsi, STATE);
mov(rdx, rax); // rax is set at the block end
// Backup rax
mov(r9, rax);
mov(rdi, ~0ULL);
sub(rsp, 16);
mov(qword [rsp], rdi);
mov(qword [rsp + 8], rdi);
mov(rdi, SIG_SETMASK);
mov(rsi, rsp);
mov(rdx, rsp);
mov(r10, 8);
mov(rax, SYS_rt_sigprocmask);
syscall();
mov(rax, r9);
}
// {rdi, rsi}
mov(rdi, STATE);
mov(rsi, rax); // rax is set at the block end
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
// Backup rax
mov(r9, rax);
mov(rdi, SIG_SETMASK);
mov(rsi, rsp);
mov(rdx, 0); // Don't care about result
mov(r10, 8);
mov(rax, SYS_rt_sigprocmask);
syscall();
// Bring stack back
add(rsp, 16);
jmp(r9);
}
else {
jmp(rax);
}
mov(rax, config.ExitFunctionLink);
call(rax);
jmp(rax);
}
{
@@ -304,7 +226,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
}
{
CallbackPtr = getCurr<JITCallback>();
CallbackPtr = getCurr<CPUBackend::JITCallback>();
push(rbx);
push(rbp);
@@ -319,7 +241,8 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
// XXX: XMM?
// Make sure to adjust the refcounter so we don't clear the cache now
add(qword STATE_PTR(CpuStateFrame, SignalHandlerRefCounter), 1);
mov(rax, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
add(dword [rax], 1);
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
@@ -327,12 +250,12 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
// Store the trampoline to the guest stack
// Guest stack is now correctly misaligned after a regular call instruction
sub(qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]), 16);
mov(rbx, qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 16);
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])]);
mov(qword [rbx], rax);
// Store RIP to the context state
mov(qword STATE_PTR(CpuStateFrame, State.rip), rsi);
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], rsi);
// Back to the loop top now
jmp(LoopTop);
@@ -341,47 +264,14 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
{
// Signal return handler
SignalHandlerReturnAddress = getCurr<uint64_t>();
ud2();
}
{
// RT Signal return handler
SignalHandlerReturnAddressRT = getCurr<uint64_t>();
ud2();
}
{
// Guest SIGILL handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGILL = getCurr<uint64_t>();
ud2();
}
{
// Guest SIGTRAP handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGTRAP = getCurr<uint64_t>();
// ud2 = SIGILL
// int3 = SIGTRAP
// hlt = SIGSEGV
int3();
}
{
// Guest SIGSEGV handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGSEGV = getCurr<uint64_t>();
// ud2 = SIGILL
// int3 = SIGTRAP
// hlt = SIGSEGV
hlt();
}
{
IntCallbackReturnAddress = getCurr<uint64_t>();
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
// using CallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
// rdi = thread
// rsi = rsp
@@ -406,101 +296,30 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
Start = reinterpret_cast<uint64_t>(getCode());
End = Start + getSize();
if (CTX->Config.BlockJITNaming()) {
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
CTX->Symbols.Register(reinterpret_cast<void*>(Start), End-Start, Name);
}
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
}
}
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline
static thread_local Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
using namespace Xbyak;
using namespace Xbyak::util;
emit.setNewBuffer(CodeBuffer, MaxGDBPauseCheckSize);
Label RunBlock;
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
emit.mov(rax, reinterpret_cast<uint64_t>(CTX));
// If the value == 0 then we don't need to stop
emit.cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
emit.je(RunBlock);
{
// Make sure RIP is syncronized to the context
emit.mov(rax, GuestRIP);
emit.mov(qword STATE_PTR(CpuStateFrame, State.rip), rax);
// Stop the thread
emit.mov(rax, qword STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
emit.jmp(rax);
}
emit.L(RunBlock);
emit.ready();
return emit.getSize();
}
size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
using namespace Xbyak;
using namespace Xbyak::util;
emit.setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
Label InlineIRData;
emit.mov(rdi, STATE);
emit.lea(rsi, ptr[rip + InlineIRData]);
emit.call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
emit.jmp(qword STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
emit.L(InlineIRData);
emit.ready();
return emit.getSize();
#if ENABLE_JITSYMBOLS
std::string Name = "Dispatch_" + std::to_string(::gettid());
CTX->Symbols.Register(Start, End-Start, Name);
#endif
}
X86Dispatcher::~X86Dispatcher() {
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
}
#ifdef _M_X86_64
void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto &Common = Thread->CurrentFrame->Pointers.Common;
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
DispatcherConfig config;
config.ExecuteBlocksWithCall = true;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddress;
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddress;
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
Common.SignalReturnHandler = SignalHandlerReturnAddress;
Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
Dispatcher = new X86Dispatcher(ctx, Thread, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
(uintptr_t&)Interpreter.CallbackReturn = IntCallbackReturnAddress;
}
// TODO: It feels wrong to initialize this way
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
}
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
return std::make_unique<X86Dispatcher>(CTX, Config);
}
#endif
}
@@ -5,24 +5,13 @@
#define XBYAK64
#include <xbyak/xbyak.h>
namespace FEXCore::Context {
struct Context;
}
namespace FEXCore::Core {
struct InternalThreadState;
}
namespace FEXCore::CPU {
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
public:
X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
virtual ~X86Dispatcher() override;
};
}
}
File diff suppressed because it is too large. Load diff
+9 -38
View File
@@ -1,13 +1,11 @@
#pragma once
#include <FEXCore/Debug/X86Tables.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <array>
#include <cstdint>
#include <utility>
#include <set>
#include <stddef.h>
#include <stack>
#include <vector>
namespace FEXCore::Context {
@@ -26,51 +24,31 @@ public:
};
Decoder(FEXCore::Context::Context *ctx);
~Decoder();
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
bool DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
std::vector<DecodedBlocks> const *GetDecodedBlocks() {
return &Blocks;
}
uint64_t DecodedMinAddress {};
uint64_t DecodedMaxAddress {~0ULL};
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
void DelayedDisownBuffer() {
PoolObject.DelayedDisownBuffer();
}
private:
// To pass any information from instruction prefixes
// down into the actual instruction handling machinery.
struct DecodedHeader {
uint8_t vvvv; // Encoded operand in a VEX prefix.
bool w; // VEX.W bit.
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
};
FEXCore::Context::Context *CTX;
const FEXCore::HLE::SyscallOSABI OSABI{};
bool DecodeInstruction(uint64_t PC);
void BranchTargetInMultiblockRange();
bool BranchTargetCanContinue(bool FinalInstruction) const;
uint8_t ReadByte();
uint8_t PeekByte(uint8_t Offset) const;
uint8_t PeekByte(uint8_t Offset);
uint64_t ReadData(uint8_t Size);
void SkipBytes(uint8_t Size) { InstructionSize += Size; }
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
bool NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
std::vector<FEXCore::X86Tables::DecodedInst> DecodedBuffer;
size_t DecodedSize {};
uint8_t const *InstStream;
@@ -87,26 +65,19 @@ private:
uint64_t MaxCondBranchBackwards {~0ULL};
uint64_t SymbolMaxAddress {};
uint64_t SymbolMinAddress {~0ULL};
uint64_t SectionMaxAddress {~0ULL};
std::vector<DecodedBlocks> Blocks;
std::set<uint64_t> BlocksToDecode;
std::set<uint64_t> HasBlocks;
std::set<uint64_t> *ExternalBranches {nullptr};
// ModRM rm decoding
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp{
const std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp {
&FEXCore::Frontend::Decoder::DecodeModRM_64,
&FEXCore::Frontend::Decoder::DecodeModRM_16,
};
const uint8_t *AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP);
FEXCORE_TELEMETRY_INIT(VEXOpTelem, TYPE_USES_VEX_OPS);
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
};
}
File diff suppressed because it is too large. Load diff
+16 -40
View File
@@ -5,23 +5,18 @@ $end_info$
*/
#pragma once
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/Event.h>
#include <mutex>
#include <thread>
#include "Interface/Context/Context.h"
#include "Common/NetStream.h"
#include <FEXCore/Utils/Threads.h>
#include <atomic>
#include <istream>
#include <memory>
#include <mutex>
#include <stdint.h>
#include <string>
namespace FEXCore {
namespace Context {
struct Context;
}
class GdbServer {
public:
GdbServer(FEXCore::Context::Context *ctx);
@@ -29,24 +24,16 @@ public:
// Public for threading
void GdbServerLoop();
void AlertLibrariesChanged() {
LibraryMapChanged = true;
}
private:
void Break(int signal);
void OpenListenSocket();
std::unique_ptr<std::iostream> OpenSocket();
void StartThread();
std::string ReadPacket(std::iostream &stream);
void SendPacket(std::ostream &stream, const std::string& packet);
void SendPacket(std::ostream &stream, std::string packet);
void SendACK(std::ostream &stream, bool NACK);
Event ThreadBreakEvent{};
void WaitForThreadWakeup();
struct HandledPacketType {
std::string Response{};
enum ResponseType {
@@ -60,20 +47,18 @@ private:
ResponseType TypeResponse{};
};
void SendPacketPair(const HandledPacketType& packetPair);
HandledPacketType ProcessPacket(const std::string &packet);
HandledPacketType handleQuery(const std::string &packet);
HandledPacketType handleXfer(const std::string &packet);
HandledPacketType handleMemory(const std::string &packet);
HandledPacketType handleV(const std::string& packet);
HandledPacketType handleThreadOp(const std::string &packet);
HandledPacketType handleBreakpoint(const std::string &packet);
void SendPacketPair(HandledPacketType packetPair);
HandledPacketType ProcessPacket(std::string &packet);
HandledPacketType handleQuery(std::string &packet);
HandledPacketType handleXfer(std::string &packet);
HandledPacketType handleMemory(std::string &packet);
HandledPacketType handleV(std::string& packet);
HandledPacketType handleThreadOp(std::string &packet);
HandledPacketType handleBreakpoint(std::string &packet);
HandledPacketType handleProgramOffsets();
HandledPacketType ThreadAction(char action, uint32_t tid);
std::string readRegs();
HandledPacketType readReg(const std::string& packet);
HandledPacketType readReg(std::string& packet);
FEXCore::Context::Context *CTX;
std::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
@@ -81,17 +66,8 @@ private:
std::mutex sendMutex;
bool SettingNoAckMode{false};
bool NoAckMode{false};
bool NonStopMode{false};
std::string ThreadString{};
std::string OSDataString{};
void buildLibraryMap();
std::atomic<bool> LibraryMapChanged = true;
std::string LibraryMapString{};
// Used to keep track of which signals to pass to the guest
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
uint32_t CurrentDebuggingThread{};
int ListenSocket{};
FEX_CONFIG_OPT(Filename, APP_FILENAME);
};
+5 -143
View File
@@ -1,7 +1,6 @@
#include "Interface/Core/CPUID.h"
#include <FEXCore/Core/HostFeatures.h>
#include "Interface/Core/HostFeatures.h"
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
#ifdef _M_ARM_64
#include "aarch64/assembler-aarch64.h"
#include "aarch64/cpu-aarch64.h"
#include "aarch64/disasm-aarch64.h"
@@ -14,151 +13,14 @@
namespace FEXCore {
// Data Zero Prohibited flag
// 0b0 = ZVA/GVA/GZVA permitted
// 0b1 = ZVA/GVA/GZVA prohibited
constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
// Log2 of the blocksize in 32-bit words
constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
#ifdef _M_ARM_64
static uint32_t GetDCZID() {
uint64_t Result{};
__asm("mrs %[Res], DCZID_EL0"
: [Res] "=r" (Result));
return Result;
}
static uint32_t GetFPCR() {
uint64_t Result{};
__asm ("mrs %[Res], FPCR"
: [Res] "=r" (Result));
return Result;
}
static void SetFPCR(uint64_t Value) {
__asm ("msr FPCR, %[Value]"
:: [Value] "r" (Value));
}
#else
static uint32_t GetDCZID() {
// Return unsupported
return DCZID_DZP_MASK;
}
#endif
HostFeatures::HostFeatures() {
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
#ifdef VIXL_SIMULATOR
auto Features = vixl::CPUFeatures::All();
#else
auto Features = vixl::CPUFeatures::InferFromOS();
#endif
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
// Only supported when FEAT_AFP is supported
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
Supports3DNow = true;
SupportsSSE4A = true;
#ifdef VIXL_SIMULATOR
// Hardcode enable SVE with 256-bit wide registers.
SupportsAVX = true;
#else
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
#endif
SupportsSHA = true;
SupportsBMI1 = true;
SupportsBMI2 = true;
SupportsCLWB = true;
if (!SupportsAtomics) {
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
}
#ifdef _M_ARM_64
// We need to get the CPU's cache line size
// We expect sane targets that have correct cacheline sizes across clusters
uint64_t CTR;
__asm volatile ("mrs %[ctr], ctr_el0"
: [ctr] "=r"(CTR));
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
ICacheLineSize = 4 << (CTR & 0xF);
// Test if this CPU supports float exception trapping by attempting to enable
// On unsupported these bits are architecturally defined as RAZ/WI
constexpr uint32_t ExceptionEnableTraps =
(1U << 8) | // Invalid Operation float exception trap enable
(1U << 9) | // Divide by zero float exception trap enable
(1U << 10) | // Overflow float exception trap enable
(1U << 11) | // Underflow float exception trap enable
(1U << 12) | // Inexact float exception trap enable
(1U << 15); // Input Denormal float exception trap enable
uint32_t OriginalFPCR = GetFPCR();
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
SetFPCR(FPCR);
FPCR = GetFPCR();
SupportsFloatExceptions = (FPCR & ExceptionEnableTraps) == ExceptionEnableTraps;
// Set FPCR back to original just in case anything changed
SetFPCR(OriginalFPCR);
auto Features = vixl::CPUFeatures::InferFromOS();
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
#endif
#endif
#if defined(_M_X86_64) && !defined(VIXL_SIMULATOR)
#ifdef _M_X86_64
Xbyak::util::Cpu Features{};
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
SupportsRCPC = true;
SupportsTSOImm9 = true;
Supports3DNow = Features.has(Xbyak::util::Cpu::t3DN) && Features.has(Xbyak::util::Cpu::tE3DN);
SupportsSSE4A = Features.has(Xbyak::util::Cpu::tSSE4a);
SupportsAVX = true;
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tCLWB);
SupportsPMULL_128Bit = Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
// xbyak doesn't know how to check for CLZero
uint32_t eax, ebx, ecx, edx;
// First ensure we support a new enough extended CPUID function range
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
if (eax >= 0x8000'0008U) {
// CLZero defined in 8000_00008_EBX[bit 0]
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
SupportsCLZERO = ebx & 1;
}
SupportsFlushInputsToZero = true;
SupportsFloatExceptions = true;
#endif
#ifdef VIXL_SIMULATOR
// simulator doesn't support dc(ZVA)
SupportsCLZERO = false;
#else
// Check if we can support cacheline clears
uint32_t DCZID = GetDCZID();
if ((DCZID & DCZID_DZP_MASK) == 0) {
uint32_t DCZID_Log2 = DCZID & DCZID_BS_MASK;
uint32_t DCZID_Bytes = (1 << DCZID_Log2) * sizeof(uint32_t);
// If the DC ZVA size matches the emulated cache line size
// This means we can use the instruction
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
}
#endif
}
}
+9
View File
@@ -0,0 +1,9 @@
#pragma once
namespace FEXCore {
class HostFeatures final {
public:
HostFeatures();
bool SupportsAES{};
};
}
File diff suppressed because it is too large. Load diff
@@ -1,777 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <FEXCore/Utils/BitUtils.h>
#include <cstdint>
namespace FEXCore::CPU {
#ifdef _M_X86_64
uint8_t AtomicFetchNeg(uint8_t *Addr) {
using Type = uint8_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
uint16_t AtomicFetchNeg(uint16_t *Addr) {
using Type = uint16_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
uint32_t AtomicFetchNeg(uint32_t *Addr) {
using Type = uint32_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
uint64_t AtomicFetchNeg(uint64_t *Addr) {
using Type = uint64_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
template<typename T>
T AtomicCompareAndSwap(T expected, T desired, T *addr)
{
std::atomic<T> *MemData = reinterpret_cast<std::atomic<T>*>(addr);
T Src1 = expected;
T Src2 = desired;
T Expected = Src1;
bool Result = MemData->compare_exchange_strong(Expected, Src2);
return Result ? Src1 : Expected;
}
template uint8_t AtomicCompareAndSwap<uint8_t>(uint8_t expected, uint8_t desired, uint8_t *addr);
template uint16_t AtomicCompareAndSwap<uint16_t>(uint16_t expected, uint16_t desired, uint16_t *addr);
template uint32_t AtomicCompareAndSwap<uint32_t>(uint32_t expected, uint32_t desired, uint32_t *addr);
template uint64_t AtomicCompareAndSwap<uint64_t>(uint64_t expected, uint64_t desired, uint64_t *addr);
#else
// Needs to match what the AArch64 JIT and unaligned signal handler expects
uint8_t AtomicFetchNeg(uint8_t *Addr) {
using Type = uint8_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxrb %w[Result], [%[Memory]];
neg %w[Tmp], %w[Result];
stlxrb %w[TmpStatus], %w[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
uint16_t AtomicFetchNeg(uint16_t *Addr) {
using Type = uint16_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxrh %w[Result], [%[Memory]];
neg %w[Tmp], %w[Result];
stlxrh %w[TmpStatus], %w[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
uint32_t AtomicFetchNeg(uint32_t *Addr) {
using Type = uint32_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxr %w[Result], [%[Memory]];
neg %w[Tmp], %w[Result];
stlxr %w[TmpStatus], %w[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
uint64_t AtomicFetchNeg(uint64_t *Addr) {
using Type = uint64_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxr %[Result], [%[Memory]];
neg %[Tmp], %[Result];
stlxr %w[TmpStatus], %[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
template<>
uint8_t AtomicCompareAndSwap(uint8_t expected, uint8_t desired, uint8_t *addr) {
using Type = uint8_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxrb %w[Tmp], [%[Memory]];
cmp %w[Tmp], %w[Expected], uxtb;
b.ne 2f;
stlxrb %w[Tmp2], %w[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %w[Result], %w[Expected];
b 3f;
2:
mov %w[Result], %w[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
template<>
uint16_t AtomicCompareAndSwap(uint16_t expected, uint16_t desired, uint16_t *addr) {
using Type = uint16_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxrh %w[Tmp], [%[Memory]];
cmp %w[Tmp], %w[Expected], uxth;
b.ne 2f;
stlxrh %w[Tmp2], %w[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %w[Result], %w[Expected];
b 3f;
2:
mov %w[Result], %w[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
template<>
uint32_t AtomicCompareAndSwap(uint32_t expected, uint32_t desired, uint32_t *addr) {
using Type = uint32_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxr %w[Tmp], [%[Memory]];
cmp %w[Tmp], %w[Expected];
b.ne 2f;
stlxr %w[Tmp2], %w[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %w[Result], %w[Expected];
b 3f;
2:
mov %w[Result], %w[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
template<>
uint64_t AtomicCompareAndSwap(uint64_t expected, uint64_t desired, uint64_t *addr) {
using Type = uint64_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxr %[Tmp], [%[Memory]];
cmp %[Tmp], %[Expected];
b.ne 2f;
stlxr %w[Tmp2], %[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %[Result], %[Expected];
b 3f;
2:
mov %[Result], %[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
#endif
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CASPair>();
// Size is the size of each pair element
switch (IROp->ElementSize) {
case 4: {
GD = AtomicCompareAndSwap(
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 8: {
std::atomic<__uint128_t> *MemData = *GetSrc<std::atomic<__uint128_t> **>(Data->SSAData, Op->Addr);
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Expected);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Desired);
__uint128_t Expected = Src1;
bool Result = MemData->compare_exchange_strong(Expected, Src2);
memcpy(GDP, Result ? &Src1 : &Expected, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", IROp->ElementSize); break;
}
}
DEF_OP(CAS) {
auto Op = IROp->C<IR::IROp_CAS>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
case 1: {
GD = AtomicCompareAndSwap(
*GetSrc<uint8_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint8_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint8_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 2: {
GD = AtomicCompareAndSwap(
*GetSrc<uint16_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint16_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint16_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 4: {
GD = AtomicCompareAndSwap(
*GetSrc<uint32_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint32_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint32_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 8: {
GD = AtomicCompareAndSwap(
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", OpSize); break;
}
}
DEF_OP(AtomicAdd) {
auto Op = IROp->C<IR::IROp_AtomicAdd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicSub) {
auto Op = IROp->C<IR::IROp_AtomicSub>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicAnd) {
auto Op = IROp->C<IR::IROp_AtomicAnd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicOr) {
auto Op = IROp->C<IR::IROp_AtomicOr>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicXor) {
auto Op = IROp->C<IR::IROp_AtomicXor>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicSwap) {
auto Op = IROp->C<IR::IROp_AtomicSwap>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchAdd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchSub) {
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchAnd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchXor) {
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchNeg) {
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
switch (IROp->Size) {
case 1: {
using Type = uint8_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
case 2: {
using Type = uint16_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
case 4: {
using Type = uint32_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
case 8: {
using Type = uint64_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,162 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include "Interface/HLE/Thunks/Thunks.h"
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <cstdint>
#include <unistd.h>
namespace FEXCore::CPU {
[[noreturn]]
static void SignalReturn(FEXCore::Core::InternalThreadState *Thread, bool RT) {
Thread->CTX->SignalThread(Thread, RT ? FEXCore::Core::SignalEvent::ReturnRT : FEXCore::Core::SignalEvent::Return);
LOGMAN_MSG_A_FMT("unreachable");
FEX_UNREACHABLE;
}
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(SignalReturn) {
auto Op = IROp->C<IR::IROp_SignalReturn>();
SignalReturn(Data->State, Op->IsRT);
}
DEF_OP(CallbackReturn) {
Data->State->CurrentFrame->Pointers.Interpreter.CallbackReturn(Data->State, Data->StackEntry);
}
DEF_OP(ExitFunction) {
auto Op = IROp->C<IR::IROp_ExitFunction>();
uint8_t OpSize = IROp->Size;
uintptr_t* ContextPtr = reinterpret_cast<uintptr_t*>(Data->State->CurrentFrame);
void *ContextData = reinterpret_cast<void*>(ContextPtr);
void *Src = GetSrc<void*>(Data->SSAData, Op->NewRIP);
memcpy(ContextData, Src, OpSize);
Data->BlockResults.Quit = true;
}
DEF_OP(Jump) {
auto Op = IROp->C<IR::IROp_Jump>();
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
const uintptr_t DataBegin = Data->CurrentIR->GetData();
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TargetBlock);
Data->BlockResults.Redo = true;
}
DEF_OP(CondJump) {
auto Op = IROp->C<IR::IROp_CondJump>();
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
const uintptr_t DataBegin = Data->CurrentIR->GetData();
bool CompResult;
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
if (Op->CompareSize == 4)
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
else
CompResult = IsConditionTrue<uint64_t, int64_t, double>(Op->Cond.Val, Src1, Src2);
if (CompResult) {
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TrueBlock);
}
else {
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->FalseBlock);
}
Data->BlockResults.Redo = true;
}
DEF_OP(Syscall) {
auto Op = IROp->C<IR::IROp_Syscall>();
FEXCore::HLE::SyscallArguments Args;
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
if (Op->Header.Args[j].IsInvalid()) break;
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
}
uint64_t Res = FEXCore::Context::HandleSyscall(Data->State->CTX->SyscallHandler, Data->State->CurrentFrame, &Args);
GD = Res;
}
DEF_OP(InlineSyscall) {
auto Op = IROp->C<IR::IROp_InlineSyscall>();
FEXCore::HLE::SyscallArguments Args;
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
if (Op->Header.Args[j].IsInvalid()) break;
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
}
// We don't want the errno handling but I also don't want to write inline ASM atm
uint64_t Res = syscall(
Op->HostSyscallNumber,
Args.Argument[0],
Args.Argument[1],
Args.Argument[2],
Args.Argument[3],
Args.Argument[4],
Args.Argument[5],
Args.Argument[6]
);
if (Res == -1) {
Res = -errno;
}
GD = Res;
}
DEF_OP(Thunk) {
auto Op = IROp->C<IR::IROp_Thunk>();
auto thunkFn = Data->State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
thunkFn(*GetSrc<void**>(Data->SSAData, Op->ArgPtr));
}
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
auto CodePtr = Data->CurrentEntry + Op->Offset;
if (memcmp((void*)CodePtr, &Op->CodeOriginalLow, Op->CodeLength) != 0) {
GD = 1;
} else {
GD = 0;
}
}
DEF_OP(ThreadRemoveCodeEntry) {
Data->State->CTX->ThreadRemoveCodeEntryFromJit(Data->State->CurrentFrame, Data->CurrentEntry);
}
DEF_OP(CPUID) {
auto Op = IROp->C<IR::IROp_CPUID>();
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
const uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Function);
const uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Leaf);
auto Results = Data->State->CTX->CPUID.RunFunction(Arg, Leaf);
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,261 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(VInsGPR) {
const auto Op = IROp->C<IR::IROp_VInsGPR>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto ElementSizeBits = ElementSize * 8;
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
const uint64_t Offset = Op->DestIdx * ElementSizeBits;
const auto InUpperLane = Offset >= SSEBitSize;
__uint128_t Mask = (1ULL << ElementSizeBits) - 1;
if (ElementSize == 8) {
Mask = ~0ULL;
}
const auto Src1 = *GetSrc<InterpVector256*>(Data->SSAData, Op->DestVector);
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
const auto Scalar = Src2 & Mask;
const auto ScaledOffset = InUpperLane ? Offset - SSEBitSize
: Offset;
// Now shift into place and set all bits but
// the ones where we're going to insert our value.
Mask <<= ScaledOffset;
Mask = ~Mask;
const auto Dst = [&] {
if (InUpperLane) {
return InterpVector256{
.Lower = Src1.Lower,
.Upper = (Src1.Upper & Mask) | (Scalar << ScaledOffset),
};
} else {
return InterpVector256{
.Lower = (Src1.Lower & Mask) | (Scalar << ScaledOffset),
.Upper = Src1.Upper,
};
}
}();
memcpy(GDP, &Dst, OpSize);
}
DEF_OP(VCastFromGPR) {
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Src), Op->Header.ElementSize);
}
DEF_OP(Float_FromGPR_S) {
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0404: { // Float <- int32_t
const float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
case 0x0408: { // Float <- int64_t
const float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
case 0x0804: { // Double <- int32_t
const double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
case 0x0808: { // Double <- int64_t
const double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
}
}
DEF_OP(Float_FToF) {
auto Op = IROp->C<IR::IROp_Float_FToF>();
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0804: { // Double <- Float
const double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Scalar);
memcpy(GDP, &Dst, 8);
break;
}
case 0x0408: { // Float <- Double
const float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Scalar);
memcpy(GDP, &Dst, 4);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
}
}
DEF_OP(Vector_SToF) {
auto Op = IROp->C<IR::IROp_Vector_SToF>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToZS) {
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToS) {
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToF) {
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint16_t ElementSize = Op->Header.ElementSize;
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (Conv) {
case 0x0804: { // Double <- float
// Only the lower elements from the source
// This uses half the source elements
uint8_t Elements = OpSize / 8;
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(double, float, Func, 0, 0)
break;
}
case 0x0408: { // Float <- Double
// Little bit tricky here
// Sometimes is used to convert from a 128bit vector register
// in to a 64bit vector register with different sized elements
// eg: %ssa5 i32v2 = Vector_FToF %ssa4 i128, #0x8
uint8_t Elements = OpSize == 8 ? 2 : OpSize / Op->SrcElementSize;
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToI) {
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func_Nearest = [](auto a) { return std::rint(a); };
const auto Func_Neg = [](auto a) { return std::floor(a); };
const auto Func_Pos = [](auto a) { return std::ceil(a); };
const auto Func_Trunc = [](auto a) { return std::trunc(a); };
const auto Func_Host = [](auto a) { return std::rint(a); };
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
}
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
}
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
}
break;
case FEXCore::IR::Round_Towards_Zero.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
}
break;
case FEXCore::IR::Round_Host.Val:
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Host)
DO_VECTOR_1SRC_OP(8, double, Func_Host)
}
break;
}
memcpy(GDP, Tmp, OpSize);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,556 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace AES {
static __uint128_t InvShiftRows(uint8_t *State) {
uint8_t Shifted[16] = {
State[0], State[13], State[10], State[7],
State[4], State[1], State[14], State[11],
State[8], State[5], State[2], State[15],
State[12], State[9], State[6], State[3],
};
__uint128_t Res{};
memcpy(&Res, Shifted, 16);
return Res;
}
static __uint128_t InvSubBytes(uint8_t *State) {
// 16x16 matrix table
static const uint8_t InvSubstitutionTable[256] = {
0x52, 0x09, 0x6a, 0xd5, 0x30, 0x36, 0xa5, 0x38, 0xbf, 0x40, 0xa3, 0x9e, 0x81, 0xf3, 0xd7, 0xfb,
0x7c, 0xe3, 0x39, 0x82, 0x9b, 0x2f, 0xff, 0x87, 0x34, 0x8e, 0x43, 0x44, 0xc4, 0xde, 0xe9, 0xcb,
0x54, 0x7b, 0x94, 0x32, 0xa6, 0xc2, 0x23, 0x3d, 0xee, 0x4c, 0x95, 0x0b, 0x42, 0xfa, 0xc3, 0x4e,
0x08, 0x2e, 0xa1, 0x66, 0x28, 0xd9, 0x24, 0xb2, 0x76, 0x5b, 0xa2, 0x49, 0x6d, 0x8b, 0xd1, 0x25,
0x72, 0xf8, 0xf6, 0x64, 0x86, 0x68, 0x98, 0x16, 0xd4, 0xa4, 0x5c, 0xcc, 0x5d, 0x65, 0xb6, 0x92,
0x6c, 0x70, 0x48, 0x50, 0xfd, 0xed, 0xb9, 0xda, 0x5e, 0x15, 0x46, 0x57, 0xa7, 0x8d, 0x9d, 0x84,
0x90, 0xd8, 0xab, 0x00, 0x8c, 0xbc, 0xd3, 0x0a, 0xf7, 0xe4, 0x58, 0x05, 0xb8, 0xb3, 0x45, 0x06,
0xd0, 0x2c, 0x1e, 0x8f, 0xca, 0x3f, 0x0f, 0x02, 0xc1, 0xaf, 0xbd, 0x03, 0x01, 0x13, 0x8a, 0x6b,
0x3a, 0x91, 0x11, 0x41, 0x4f, 0x67, 0xdc, 0xea, 0x97, 0xf2, 0xcf, 0xce, 0xf0, 0xb4, 0xe6, 0x73,
0x96, 0xac, 0x74, 0x22, 0xe7, 0xad, 0x35, 0x85, 0xe2, 0xf9, 0x37, 0xe8, 0x1c, 0x75, 0xdf, 0x6e,
0x47, 0xf1, 0x1a, 0x71, 0x1d, 0x29, 0xc5, 0x89, 0x6f, 0xb7, 0x62, 0x0e, 0xaa, 0x18, 0xbe, 0x1b,
0xfc, 0x56, 0x3e, 0x4b, 0xc6, 0xd2, 0x79, 0x20, 0x9a, 0xdb, 0xc0, 0xfe, 0x78, 0xcd, 0x5a, 0xf4,
0x1f, 0xdd, 0xa8, 0x33, 0x88, 0x07, 0xc7, 0x31, 0xb1, 0x12, 0x10, 0x59, 0x27, 0x80, 0xec, 0x5f,
0x60, 0x51, 0x7f, 0xa9, 0x19, 0xb5, 0x4a, 0x0d, 0x2d, 0xe5, 0x7a, 0x9f, 0x93, 0xc9, 0x9c, 0xef,
0xa0, 0xe0, 0x3b, 0x4d, 0xae, 0x2a, 0xf5, 0xb0, 0xc8, 0xeb, 0xbb, 0x3c, 0x83, 0x53, 0x99, 0x61,
0x17, 0x2b, 0x04, 0x7e, 0xba, 0x77, 0xd6, 0x26, 0xe1, 0x69, 0x14, 0x63, 0x55, 0x21, 0x0c, 0x7d,
};
// Uses a byte substitution table with a constant set of values
// Needs to do a table look up
uint8_t Substituted[16];
for (size_t i = 0; i < 16; ++i) {
Substituted[i] = InvSubstitutionTable[State[i]];
}
__uint128_t Res{};
memcpy(&Res, Substituted, 16);
return Res;
}
static __uint128_t ShiftRows(uint8_t *State) {
uint8_t Shifted[16] = {
State[0], State[5], State[10], State[15],
State[4], State[9], State[14], State[3],
State[8], State[13], State[2], State[7],
State[12], State[1], State[6], State[11],
};
__uint128_t Res{};
memcpy(&Res, Shifted, 16);
return Res;
}
static __uint128_t SubBytes(uint8_t *State, size_t Bytes) {
// 16x16 matrix table
static const uint8_t SubstitutionTable[256] = {
0x63, 0x7c, 0x77, 0x7b, 0xf2, 0x6b, 0x6f, 0xc5, 0x30, 0x01, 0x67, 0x2b, 0xfe, 0xd7, 0xab, 0x76,
0xca, 0x82, 0xc9, 0x7d, 0xfa, 0x59, 0x47, 0xf0, 0xad, 0xd4, 0xa2, 0xaf, 0x9c, 0xa4, 0x72, 0xc0,
0xb7, 0xfd, 0x93, 0x26, 0x36, 0x3f, 0xf7, 0xcc, 0x34, 0xa5, 0xe5, 0xf1, 0x71, 0xd8, 0x31, 0x15,
0x04, 0xc7, 0x23, 0xc3, 0x18, 0x96, 0x05, 0x9a, 0x07, 0x12, 0x80, 0xe2, 0xeb, 0x27, 0xb2, 0x75,
0x09, 0x83, 0x2c, 0x1a, 0x1b, 0x6e, 0x5a, 0xa0, 0x52, 0x3b, 0xd6, 0xb3, 0x29, 0xe3, 0x2f, 0x84,
0x53, 0xd1, 0x00, 0xed, 0x20, 0xfc, 0xb1, 0x5b, 0x6a, 0xcb, 0xbe, 0x39, 0x4a, 0x4c, 0x58, 0xcf,
0xd0, 0xef, 0xaa, 0xfb, 0x43, 0x4d, 0x33, 0x85, 0x45, 0xf9, 0x02, 0x7f, 0x50, 0x3c, 0x9f, 0xa8,
0x51, 0xa3, 0x40, 0x8f, 0x92, 0x9d, 0x38, 0xf5, 0xbc, 0xb6, 0xda, 0x21, 0x10, 0xff, 0xf3, 0xd2,
0xcd, 0x0c, 0x13, 0xec, 0x5f, 0x97, 0x44, 0x17, 0xc4, 0xa7, 0x7e, 0x3d, 0x64, 0x5d, 0x19, 0x73,
0x60, 0x81, 0x4f, 0xdc, 0x22, 0x2a, 0x90, 0x88, 0x46, 0xee, 0xb8, 0x14, 0xde, 0x5e, 0x0b, 0xdb,
0xe0, 0x32, 0x3a, 0x0a, 0x49, 0x06, 0x24, 0x5c, 0xc2, 0xd3, 0xac, 0x62, 0x91, 0x95, 0xe4, 0x79,
0xe7, 0xc8, 0x37, 0x6d, 0x8d, 0xd5, 0x4e, 0xa9, 0x6c, 0x56, 0xf4, 0xea, 0x65, 0x7a, 0xae, 0x08,
0xba, 0x78, 0x25, 0x2e, 0x1c, 0xa6, 0xb4, 0xc6, 0xe8, 0xdd, 0x74, 0x1f, 0x4b, 0xbd, 0x8b, 0x8a,
0x70, 0x3e, 0xb5, 0x66, 0x48, 0x03, 0xf6, 0x0e, 0x61, 0x35, 0x57, 0xb9, 0x86, 0xc1, 0x1d, 0x9e,
0xe1, 0xf8, 0x98, 0x11, 0x69, 0xd9, 0x8e, 0x94, 0x9b, 0x1e, 0x87, 0xe9, 0xce, 0x55, 0x28, 0xdf,
0x8c, 0xa1, 0x89, 0x0d, 0xbf, 0xe6, 0x42, 0x68, 0x41, 0x99, 0x2d, 0x0f, 0xb0, 0x54, 0xbb, 0x16,
};
// Uses a byte substitution table with a constant set of values
// Needs to do a table look up
uint8_t Substituted[16];
Bytes = std::min(Bytes, (size_t)16);
for (size_t i = 0; i < Bytes; ++i) {
Substituted[i] = SubstitutionTable[State[i]];
}
__uint128_t Res{};
memcpy(&Res, Substituted, Bytes);
return Res;
}
static uint8_t FFMul02(uint8_t in) {
static const uint8_t FFMul02[256] = {
0x00, 0x02, 0x04, 0x06, 0x08, 0x0a, 0x0c, 0x0e, 0x10, 0x12, 0x14, 0x16, 0x18, 0x1a, 0x1c, 0x1e,
0x20, 0x22, 0x24, 0x26, 0x28, 0x2a, 0x2c, 0x2e, 0x30, 0x32, 0x34, 0x36, 0x38, 0x3a, 0x3c, 0x3e,
0x40, 0x42, 0x44, 0x46, 0x48, 0x4a, 0x4c, 0x4e, 0x50, 0x52, 0x54, 0x56, 0x58, 0x5a, 0x5c, 0x5e,
0x60, 0x62, 0x64, 0x66, 0x68, 0x6a, 0x6c, 0x6e, 0x70, 0x72, 0x74, 0x76, 0x78, 0x7a, 0x7c, 0x7e,
0x80, 0x82, 0x84, 0x86, 0x88, 0x8a, 0x8c, 0x8e, 0x90, 0x92, 0x94, 0x96, 0x98, 0x9a, 0x9c, 0x9e,
0xa0, 0xa2, 0xa4, 0xa6, 0xa8, 0xaa, 0xac, 0xae, 0xb0, 0xb2, 0xb4, 0xb6, 0xb8, 0xba, 0xbc, 0xbe,
0xc0, 0xc2, 0xc4, 0xc6, 0xc8, 0xca, 0xcc, 0xce, 0xd0, 0xd2, 0xd4, 0xd6, 0xd8, 0xda, 0xdc, 0xde,
0xe0, 0xe2, 0xe4, 0xe6, 0xe8, 0xea, 0xec, 0xee, 0xf0, 0xf2, 0xf4, 0xf6, 0xf8, 0xfa, 0xfc, 0xfe,
0x1b, 0x19, 0x1f, 0x1d, 0x13, 0x11, 0x17, 0x15, 0x0b, 0x09, 0x0f, 0x0d, 0x03, 0x01, 0x07, 0x05,
0x3b, 0x39, 0x3f, 0x3d, 0x33, 0x31, 0x37, 0x35, 0x2b, 0x29, 0x2f, 0x2d, 0x23, 0x21, 0x27, 0x25,
0x5b, 0x59, 0x5f, 0x5d, 0x53, 0x51, 0x57, 0x55, 0x4b, 0x49, 0x4f, 0x4d, 0x43, 0x41, 0x47, 0x45,
0x7b, 0x79, 0x7f, 0x7d, 0x73, 0x71, 0x77, 0x75, 0x6b, 0x69, 0x6f, 0x6d, 0x63, 0x61, 0x67, 0x65,
0x9b, 0x99, 0x9f, 0x9d, 0x93, 0x91, 0x97, 0x95, 0x8b, 0x89, 0x8f, 0x8d, 0x83, 0x81, 0x87, 0x85,
0xbb, 0xb9, 0xbf, 0xbd, 0xb3, 0xb1, 0xb7, 0xb5, 0xab, 0xa9, 0xaf, 0xad, 0xa3, 0xa1, 0xa7, 0xa5,
0xdb, 0xd9, 0xdf, 0xdd, 0xd3, 0xd1, 0xd7, 0xd5, 0xcb, 0xc9, 0xcf, 0xcd, 0xc3, 0xc1, 0xc7, 0xc5,
0xfb, 0xf9, 0xff, 0xfd, 0xf3, 0xf1, 0xf7, 0xf5, 0xeb, 0xe9, 0xef, 0xed, 0xe3, 0xe1, 0xe7, 0xe5,
};
return FFMul02[in];
}
static uint8_t FFMul03(uint8_t in) {
static const uint8_t FFMul03[256] = {
0x00, 0x03, 0x06, 0x05, 0x0c, 0x0f, 0x0a, 0x09, 0x18, 0x1b, 0x1e, 0x1d, 0x14, 0x17, 0x12, 0x11,
0x30, 0x33, 0x36, 0x35, 0x3c, 0x3f, 0x3a, 0x39, 0x28, 0x2b, 0x2e, 0x2d, 0x24, 0x27, 0x22, 0x21,
0x60, 0x63, 0x66, 0x65, 0x6c, 0x6f, 0x6a, 0x69, 0x78, 0x7b, 0x7e, 0x7d, 0x74, 0x77, 0x72, 0x71,
0x50, 0x53, 0x56, 0x55, 0x5c, 0x5f, 0x5a, 0x59, 0x48, 0x4b, 0x4e, 0x4d, 0x44, 0x47, 0x42, 0x41,
0xc0, 0xc3, 0xc6, 0xc5, 0xcc, 0xcf, 0xca, 0xc9, 0xd8, 0xdb, 0xde, 0xdd, 0xd4, 0xd7, 0xd2, 0xd1,
0xf0, 0xf3, 0xf6, 0xf5, 0xfc, 0xff, 0xfa, 0xf9, 0xe8, 0xeb, 0xee, 0xed, 0xe4, 0xe7, 0xe2, 0xe1,
0xa0, 0xa3, 0xa6, 0xa5, 0xac, 0xaf, 0xaa, 0xa9, 0xb8, 0xbb, 0xbe, 0xbd, 0xb4, 0xb7, 0xb2, 0xb1,
0x90, 0x93, 0x96, 0x95, 0x9c, 0x9f, 0x9a, 0x99, 0x88, 0x8b, 0x8e, 0x8d, 0x84, 0x87, 0x82, 0x81,
0x9b, 0x98, 0x9d, 0x9e, 0x97, 0x94, 0x91, 0x92, 0x83, 0x80, 0x85, 0x86, 0x8f, 0x8c, 0x89, 0x8a,
0xab, 0xa8, 0xad, 0xae, 0xa7, 0xa4, 0xa1, 0xa2, 0xb3, 0xb0, 0xb5, 0xb6, 0xbf, 0xbc, 0xb9, 0xba,
0xfb, 0xf8, 0xfd, 0xfe, 0xf7, 0xf4, 0xf1, 0xf2, 0xe3, 0xe0, 0xe5, 0xe6, 0xef, 0xec, 0xe9, 0xea,
0xcb, 0xc8, 0xcd, 0xce, 0xc7, 0xc4, 0xc1, 0xc2, 0xd3, 0xd0, 0xd5, 0xd6, 0xdf, 0xdc, 0xd9, 0xda,
0x5b, 0x58, 0x5d, 0x5e, 0x57, 0x54, 0x51, 0x52, 0x43, 0x40, 0x45, 0x46, 0x4f, 0x4c, 0x49, 0x4a,
0x6b, 0x68, 0x6d, 0x6e, 0x67, 0x64, 0x61, 0x62, 0x73, 0x70, 0x75, 0x76, 0x7f, 0x7c, 0x79, 0x7a,
0x3b, 0x38, 0x3d, 0x3e, 0x37, 0x34, 0x31, 0x32, 0x23, 0x20, 0x25, 0x26, 0x2f, 0x2c, 0x29, 0x2a,
0x0b, 0x08, 0x0d, 0x0e, 0x07, 0x04, 0x01, 0x02, 0x13, 0x10, 0x15, 0x16, 0x1f, 0x1c, 0x19, 0x1a,
};
return FFMul03[in];
}
static __uint128_t MixColumns(uint8_t *State) {
uint8_t In0[16] = {
State[0], State[4], State[8], State[12],
State[1], State[5], State[9], State[13],
State[2], State[6], State[10], State[14],
State[3], State[7], State[11], State[15],
};
uint8_t Out0[4]{};
uint8_t Out1[4]{};
uint8_t Out2[4]{};
uint8_t Out3[4]{};
for (size_t i = 0; i < 4; ++i) {
Out0[i] = FFMul02(In0[0 + i]) ^ FFMul03(In0[4 + i]) ^ In0[8 + i] ^ In0[12 + i];
Out1[i] = In0[0 + i] ^ FFMul02(In0[4 + i]) ^ FFMul03(In0[8 + i]) ^ In0[12 + i];
Out2[i] = In0[0 + i] ^ In0[4 + i] ^ FFMul02(In0[8 + i]) ^ FFMul03(In0[12 + i]);
Out3[i] = FFMul03(In0[0 + i]) ^ In0[4 + i] ^ In0[8 + i] ^ FFMul02(In0[12 + i]);
}
uint8_t OutArray[16] = {
Out0[0], Out1[0], Out2[0], Out3[0],
Out0[1], Out1[1], Out2[1], Out3[1],
Out0[2], Out1[2], Out2[2], Out3[2],
Out0[3], Out1[3], Out2[3], Out3[3],
};
__uint128_t Res{};
memcpy(&Res, OutArray, 16);
return Res;
}
static uint8_t FFMul09(uint8_t in) {
static const uint8_t FFMul09[256] = {
0x00, 0x09, 0x12, 0x1b, 0x24, 0x2d, 0x36, 0x3f, 0x48, 0x41, 0x5a, 0x53, 0x6c, 0x65, 0x7e, 0x77,
0x90, 0x99, 0x82, 0x8b, 0xb4, 0xbd, 0xa6, 0xaf, 0xd8, 0xd1, 0xca, 0xc3, 0xfc, 0xf5, 0xee, 0xe7,
0x3b, 0x32, 0x29, 0x20, 0x1f, 0x16, 0x0d, 0x04, 0x73, 0x7a, 0x61, 0x68, 0x57, 0x5e, 0x45, 0x4c,
0xab, 0xa2, 0xb9, 0xb0, 0x8f, 0x86, 0x9d, 0x94, 0xe3, 0xea, 0xf1, 0xf8, 0xc7, 0xce, 0xd5, 0xdc,
0x76, 0x7f, 0x64, 0x6d, 0x52, 0x5b, 0x40, 0x49, 0x3e, 0x37, 0x2c, 0x25, 0x1a, 0x13, 0x08, 0x01,
0xe6, 0xef, 0xf4, 0xfd, 0xc2, 0xcb, 0xd0, 0xd9, 0xae, 0xa7, 0xbc, 0xb5, 0x8a, 0x83, 0x98, 0x91,
0x4d, 0x44, 0x5f, 0x56, 0x69, 0x60, 0x7b, 0x72, 0x05, 0x0c, 0x17, 0x1e, 0x21, 0x28, 0x33, 0x3a,
0xdd, 0xd4, 0xcf, 0xc6, 0xf9, 0xf0, 0xeb, 0xe2, 0x95, 0x9c, 0x87, 0x8e, 0xb1, 0xb8, 0xa3, 0xaa,
0xec, 0xe5, 0xfe, 0xf7, 0xc8, 0xc1, 0xda, 0xd3, 0xa4, 0xad, 0xb6, 0xbf, 0x80, 0x89, 0x92, 0x9b,
0x7c, 0x75, 0x6e, 0x67, 0x58, 0x51, 0x4a, 0x43, 0x34, 0x3d, 0x26, 0x2f, 0x10, 0x19, 0x02, 0x0b,
0xd7, 0xde, 0xc5, 0xcc, 0xf3, 0xfa, 0xe1, 0xe8, 0x9f, 0x96, 0x8d, 0x84, 0xbb, 0xb2, 0xa9, 0xa0,
0x47, 0x4e, 0x55, 0x5c, 0x63, 0x6a, 0x71, 0x78, 0x0f, 0x06, 0x1d, 0x14, 0x2b, 0x22, 0x39, 0x30,
0x9a, 0x93, 0x88, 0x81, 0xbe, 0xb7, 0xac, 0xa5, 0xd2, 0xdb, 0xc0, 0xc9, 0xf6, 0xff, 0xe4, 0xed,
0x0a, 0x03, 0x18, 0x11, 0x2e, 0x27, 0x3c, 0x35, 0x42, 0x4b, 0x50, 0x59, 0x66, 0x6f, 0x74, 0x7d,
0xa1, 0xa8, 0xb3, 0xba, 0x85, 0x8c, 0x97, 0x9e, 0xe9, 0xe0, 0xfb, 0xf2, 0xcd, 0xc4, 0xdf, 0xd6,
0x31, 0x38, 0x23, 0x2a, 0x15, 0x1c, 0x07, 0x0e, 0x79, 0x70, 0x6b, 0x62, 0x5d, 0x54, 0x4f, 0x46,
};
return FFMul09[in];
}
static uint8_t FFMul0B(uint8_t in) {
static const uint8_t FFMul0B[256] = {
0x00, 0x0b, 0x16, 0x1d, 0x2c, 0x27, 0x3a, 0x31, 0x58, 0x53, 0x4e, 0x45, 0x74, 0x7f, 0x62, 0x69,
0xb0, 0xbb, 0xa6, 0xad, 0x9c, 0x97, 0x8a, 0x81, 0xe8, 0xe3, 0xfe, 0xf5, 0xc4, 0xcf, 0xd2, 0xd9,
0x7b, 0x70, 0x6d, 0x66, 0x57, 0x5c, 0x41, 0x4a, 0x23, 0x28, 0x35, 0x3e, 0x0f, 0x04, 0x19, 0x12,
0xcb, 0xc0, 0xdd, 0xd6, 0xe7, 0xec, 0xf1, 0xfa, 0x93, 0x98, 0x85, 0x8e, 0xbf, 0xb4, 0xa9, 0xa2,
0xf6, 0xfd, 0xe0, 0xeb, 0xda, 0xd1, 0xcc, 0xc7, 0xae, 0xa5, 0xb8, 0xb3, 0x82, 0x89, 0x94, 0x9f,
0x46, 0x4d, 0x50, 0x5b, 0x6a, 0x61, 0x7c, 0x77, 0x1e, 0x15, 0x08, 0x03, 0x32, 0x39, 0x24, 0x2f,
0x8d, 0x86, 0x9b, 0x90, 0xa1, 0xaa, 0xb7, 0xbc, 0xd5, 0xde, 0xc3, 0xc8, 0xf9, 0xf2, 0xef, 0xe4,
0x3d, 0x36, 0x2b, 0x20, 0x11, 0x1a, 0x07, 0x0c, 0x65, 0x6e, 0x73, 0x78, 0x49, 0x42, 0x5f, 0x54,
0xf7, 0xfc, 0xe1, 0xea, 0xdb, 0xd0, 0xcd, 0xc6, 0xaf, 0xa4, 0xb9, 0xb2, 0x83, 0x88, 0x95, 0x9e,
0x47, 0x4c, 0x51, 0x5a, 0x6b, 0x60, 0x7d, 0x76, 0x1f, 0x14, 0x09, 0x02, 0x33, 0x38, 0x25, 0x2e,
0x8c, 0x87, 0x9a, 0x91, 0xa0, 0xab, 0xb6, 0xbd, 0xd4, 0xdf, 0xc2, 0xc9, 0xf8, 0xf3, 0xee, 0xe5,
0x3c, 0x37, 0x2a, 0x21, 0x10, 0x1b, 0x06, 0x0d, 0x64, 0x6f, 0x72, 0x79, 0x48, 0x43, 0x5e, 0x55,
0x01, 0x0a, 0x17, 0x1c, 0x2d, 0x26, 0x3b, 0x30, 0x59, 0x52, 0x4f, 0x44, 0x75, 0x7e, 0x63, 0x68,
0xb1, 0xba, 0xa7, 0xac, 0x9d, 0x96, 0x8b, 0x80, 0xe9, 0xe2, 0xff, 0xf4, 0xc5, 0xce, 0xd3, 0xd8,
0x7a, 0x71, 0x6c, 0x67, 0x56, 0x5d, 0x40, 0x4b, 0x22, 0x29, 0x34, 0x3f, 0x0e, 0x05, 0x18, 0x13,
0xca, 0xc1, 0xdc, 0xd7, 0xe6, 0xed, 0xf0, 0xfb, 0x92, 0x99, 0x84, 0x8f, 0xbe, 0xb5, 0xa8, 0xa3,
};
return FFMul0B[in];
}
static uint8_t FFMul0D(uint8_t in) {
static const uint8_t FFMul0D[256] = {
0x00, 0x0d, 0x1a, 0x17, 0x34, 0x39, 0x2e, 0x23, 0x68, 0x65, 0x72, 0x7f, 0x5c, 0x51, 0x46, 0x4b,
0xd0, 0xdd, 0xca, 0xc7, 0xe4, 0xe9, 0xfe, 0xf3, 0xb8, 0xb5, 0xa2, 0xaf, 0x8c, 0x81, 0x96, 0x9b,
0xbb, 0xb6, 0xa1, 0xac, 0x8f, 0x82, 0x95, 0x98, 0xd3, 0xde, 0xc9, 0xc4, 0xe7, 0xea, 0xfd, 0xf0,
0x6b, 0x66, 0x71, 0x7c, 0x5f, 0x52, 0x45, 0x48, 0x03, 0x0e, 0x19, 0x14, 0x37, 0x3a, 0x2d, 0x20,
0x6d, 0x60, 0x77, 0x7a, 0x59, 0x54, 0x43, 0x4e, 0x05, 0x08, 0x1f, 0x12, 0x31, 0x3c, 0x2b, 0x26,
0xbd, 0xb0, 0xa7, 0xaa, 0x89, 0x84, 0x93, 0x9e, 0xd5, 0xd8, 0xcf, 0xc2, 0xe1, 0xec, 0xfb, 0xf6,
0xd6, 0xdb, 0xcc, 0xc1, 0xe2, 0xef, 0xf8, 0xf5, 0xbe, 0xb3, 0xa4, 0xa9, 0x8a, 0x87, 0x90, 0x9d,
0x06, 0x0b, 0x1c, 0x11, 0x32, 0x3f, 0x28, 0x25, 0x6e, 0x63, 0x74, 0x79, 0x5a, 0x57, 0x40, 0x4d,
0xda, 0xd7, 0xc0, 0xcd, 0xee, 0xe3, 0xf4, 0xf9, 0xb2, 0xbf, 0xa8, 0xa5, 0x86, 0x8b, 0x9c, 0x91,
0x0a, 0x07, 0x10, 0x1d, 0x3e, 0x33, 0x24, 0x29, 0x62, 0x6f, 0x78, 0x75, 0x56, 0x5b, 0x4c, 0x41,
0x61, 0x6c, 0x7b, 0x76, 0x55, 0x58, 0x4f, 0x42, 0x09, 0x04, 0x13, 0x1e, 0x3d, 0x30, 0x27, 0x2a,
0xb1, 0xbc, 0xab, 0xa6, 0x85, 0x88, 0x9f, 0x92, 0xd9, 0xd4, 0xc3, 0xce, 0xed, 0xe0, 0xf7, 0xfa,
0xb7, 0xba, 0xad, 0xa0, 0x83, 0x8e, 0x99, 0x94, 0xdf, 0xd2, 0xc5, 0xc8, 0xeb, 0xe6, 0xf1, 0xfc,
0x67, 0x6a, 0x7d, 0x70, 0x53, 0x5e, 0x49, 0x44, 0x0f, 0x02, 0x15, 0x18, 0x3b, 0x36, 0x21, 0x2c,
0x0c, 0x01, 0x16, 0x1b, 0x38, 0x35, 0x22, 0x2f, 0x64, 0x69, 0x7e, 0x73, 0x50, 0x5d, 0x4a, 0x47,
0xdc, 0xd1, 0xc6, 0xcb, 0xe8, 0xe5, 0xf2, 0xff, 0xb4, 0xb9, 0xae, 0xa3, 0x80, 0x8d, 0x9a, 0x97,
};
return FFMul0D[in];
}
static uint8_t FFMul0E(uint8_t in) {
static const uint8_t FFMul0E[256] = {
0x00, 0x0e, 0x1c, 0x12, 0x38, 0x36, 0x24, 0x2a, 0x70, 0x7e, 0x6c, 0x62, 0x48, 0x46, 0x54, 0x5a,
0xe0, 0xee, 0xfc, 0xf2, 0xd8, 0xd6, 0xc4, 0xca, 0x90, 0x9e, 0x8c, 0x82, 0xa8, 0xa6, 0xb4, 0xba,
0xdb, 0xd5, 0xc7, 0xc9, 0xe3, 0xed, 0xff, 0xf1, 0xab, 0xa5, 0xb7, 0xb9, 0x93, 0x9d, 0x8f, 0x81,
0x3b, 0x35, 0x27, 0x29, 0x03, 0x0d, 0x1f, 0x11, 0x4b, 0x45, 0x57, 0x59, 0x73, 0x7d, 0x6f, 0x61,
0xad, 0xa3, 0xb1, 0xbf, 0x95, 0x9b, 0x89, 0x87, 0xdd, 0xd3, 0xc1, 0xcf, 0xe5, 0xeb, 0xf9, 0xf7,
0x4d, 0x43, 0x51, 0x5f, 0x75, 0x7b, 0x69, 0x67, 0x3d, 0x33, 0x21, 0x2f, 0x05, 0x0b, 0x19, 0x17,
0x76, 0x78, 0x6a, 0x64, 0x4e, 0x40, 0x52, 0x5c, 0x06, 0x08, 0x1a, 0x14, 0x3e, 0x30, 0x22, 0x2c,
0x96, 0x98, 0x8a, 0x84, 0xae, 0xa0, 0xb2, 0xbc, 0xe6, 0xe8, 0xfa, 0xf4, 0xde, 0xd0, 0xc2, 0xcc,
0x41, 0x4f, 0x5d, 0x53, 0x79, 0x77, 0x65, 0x6b, 0x31, 0x3f, 0x2d, 0x23, 0x09, 0x07, 0x15, 0x1b,
0xa1, 0xaf, 0xbd, 0xb3, 0x99, 0x97, 0x85, 0x8b, 0xd1, 0xdf, 0xcd, 0xc3, 0xe9, 0xe7, 0xf5, 0xfb,
0x9a, 0x94, 0x86, 0x88, 0xa2, 0xac, 0xbe, 0xb0, 0xea, 0xe4, 0xf6, 0xf8, 0xd2, 0xdc, 0xce, 0xc0,
0x7a, 0x74, 0x66, 0x68, 0x42, 0x4c, 0x5e, 0x50, 0x0a, 0x04, 0x16, 0x18, 0x32, 0x3c, 0x2e, 0x20,
0xec, 0xe2, 0xf0, 0xfe, 0xd4, 0xda, 0xc8, 0xc6, 0x9c, 0x92, 0x80, 0x8e, 0xa4, 0xaa, 0xb8, 0xb6,
0x0c, 0x02, 0x10, 0x1e, 0x34, 0x3a, 0x28, 0x26, 0x7c, 0x72, 0x60, 0x6e, 0x44, 0x4a, 0x58, 0x56,
0x37, 0x39, 0x2b, 0x25, 0x0f, 0x01, 0x13, 0x1d, 0x47, 0x49, 0x5b, 0x55, 0x7f, 0x71, 0x63, 0x6d,
0xd7, 0xd9, 0xcb, 0xc5, 0xef, 0xe1, 0xf3, 0xfd, 0xa7, 0xa9, 0xbb, 0xb5, 0x9f, 0x91, 0x83, 0x8d,
};
return FFMul0E[in];
}
static __uint128_t InvMixColumns(uint8_t *State) {
uint8_t In0[16] = {
State[0], State[4], State[8], State[12],
State[1], State[5], State[9], State[13],
State[2], State[6], State[10], State[14],
State[3], State[7], State[11], State[15],
};
uint8_t Out0[4]{};
uint8_t Out1[4]{};
uint8_t Out2[4]{};
uint8_t Out3[4]{};
for (size_t i = 0; i < 4; ++i) {
Out0[i] = FFMul0E(In0[0 + i]) ^ FFMul0B(In0[4 + i]) ^ FFMul0D(In0[8 + i]) ^ FFMul09(In0[12 + i]);
Out1[i] = FFMul09(In0[0 + i]) ^ FFMul0E(In0[4 + i]) ^ FFMul0B(In0[8 + i]) ^ FFMul0D(In0[12 + i]);
Out2[i] = FFMul0D(In0[0 + i]) ^ FFMul09(In0[4 + i]) ^ FFMul0E(In0[8 + i]) ^ FFMul0B(In0[12 + i]);
Out3[i] = FFMul0B(In0[0 + i]) ^ FFMul0D(In0[4 + i]) ^ FFMul09(In0[8 + i]) ^ FFMul0E(In0[12 + i]);
}
uint8_t OutArray[16] = {
Out0[0], Out1[0], Out2[0], Out3[0],
Out0[1], Out1[1], Out2[1], Out3[1],
Out0[2], Out1[2], Out2[2], Out3[2],
Out0[3], Out1[3], Out2[3], Out3[3],
};
__uint128_t Res{};
memcpy(&Res, OutArray, 16);
return Res;
}
}
namespace CRC32 {
// CRC32 per byte lookup table.
constexpr std::array<uint32_t, 256> CRC32CTable = []() consteval {
std::array<uint32_t, 256> Table{};
// Clang 11.x doesn't support bitreverse as a consteval
// constexpr uint32_t Polynomial = 0x1EDC6F41;
constexpr uint32_t PolynomialRev = 0x82F63B78; //__builtin_bitreverse32(Polynomial);
for (size_t Char = 0; Char < std::size(Table); ++Char) {
uint32_t CurrentChar = Char;
for (size_t i = 0; i < 8; ++i) {
if (CurrentChar & 1) {
CurrentChar = (CurrentChar >> 1) ^ PolynomialRev;
}
else {
CurrentChar >>= 1;
}
}
Table[Char] = CurrentChar;
}
return Table;
}();
uint32_t crc32cb(uint32_t Accumulator, uint8_t data) {
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ data] ^ Accumulator >> 8;
return Accumulator;
}
uint32_t crc32ch(uint32_t Accumulator, uint16_t data) {
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
return Accumulator;
}
uint32_t crc32cw(uint32_t Accumulator, uint32_t data) {
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
return Accumulator;
}
uint32_t crc32cx(uint32_t Accumulator, uint64_t data) {
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 32) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 40) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 48) & 0xFF)] ^ Accumulator >> 8;
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 56) & 0xFF)] ^ Accumulator >> 8;
return Accumulator;
}
}
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(AESImc) {
auto Op = IROp->C<IR::IROp_VAESImc>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
// Pseudo-code
// Dst = InvMixColumns(STATE)
__uint128_t Tmp{};
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Src1));
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESEnc) {
auto Op = IROp->C<IR::IROp_VAESEnc>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = ShiftRows(STATE)
// STATE = SubBytes(STATE)
// STATE = MixColumns(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
Tmp = AES::MixColumns(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESEncLast) {
auto Op = IROp->C<IR::IROp_VAESEncLast>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = ShiftRows(STATE)
// STATE = SubBytes(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESDec) {
auto Op = IROp->C<IR::IROp_VAESDec>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = InvShiftRows(STATE)
// STATE = InvSubBytes(STATE)
// STATE = InvMixColumns(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESDecLast) {
auto Op = IROp->C<IR::IROp_VAESDecLast>();
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = InvShiftRows(STATE)
// STATE = InvSubBytes(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESKeyGenAssist) {
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Src);
// Pseudo-code
// X3 = Src1[127:96]
// X2 = Src1[95:64]
// X1 = Src1[63:32]
// X0 = Src1[31:30]
// RCON = (Zext)rcon
// Dest[31:0] = SubWord(X1)
// Dest[63:32] = RotWord(SubWord(X1)) XOR RCON
// Dest[95:64] = SubWord(X3)
// Dest[127:96] = RotWord(SubWord(X3)) XOR RCON
__uint128_t Tmp{};
uint32_t X1{};
uint32_t X3{};
memcpy(&X1, &Src1[4], 4);
memcpy(&X3, &Src1[12], 4);
uint32_t SubWord_X1 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X1), 4);
uint32_t SubWord_X3 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X3), 4);
auto Ror = [] (auto In, auto R) {
auto RotateMask = sizeof(In) * 8 - 1;
R &= RotateMask;
return (In >> R) | (In << (sizeof(In) * 8 - R));
};
uint32_t Rot_X1 = Ror(SubWord_X1, 8);
uint32_t Rot_X3 = Ror(SubWord_X3, 8);
Tmp = Rot_X3 ^ Op->RCON;
Tmp <<= 32;
Tmp |= SubWord_X3;
Tmp <<= 32;
Tmp |= Rot_X1 ^ Op->RCON;
Tmp <<= 32;
Tmp |= SubWord_X1;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(CRC32) {
auto Op = IROp->C<IR::IROp_CRC32>();
uint32_t Src1 = *GetSrc<uint32_t*>(Data->SSAData, Op->Src1);
uint8_t *Src2 = GetSrc<uint8_t*>(Data->SSAData, Op->Src2);
uint32_t Tmp{};
switch (Op->SrcSize) {
case 1:
Tmp = CRC32::crc32cb(Src1, *(uint8_t*)Src2);
break;
case 2:
Tmp = CRC32::crc32ch(Src1, *(uint16_t*)Src2);
break;
case 4:
Tmp = CRC32::crc32cw(Src1, *(uint32_t*)Src2);
break;
case 8:
Tmp = CRC32::crc32cx(Src1, *(uint64_t*)Src2);
break;
default:
LOGMAN_MSG_A_FMT("Unknown CRC32C size: {}", Op->SrcSize);
break;
}
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(PCLMUL) {
auto Op = IROp->C<IR::IROp_PCLMUL>();
const auto Selector = Op->Selector;
auto* Dst = GetDest<uint64_t*>(Data->SSAData, Node);
auto* Src1 = GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
auto* Src2 = GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
const uint64_t TMP1 = (Selector & 0x01) == 0 ? Src1[0] : Src1[1];
const uint64_t TMP2 = (Selector & 0x10) == 0 ? Src2[0] : Src2[1];
const auto make_lo = [](uint64_t lhs, uint64_t rhs) {
uint64_t result = 0;
for (size_t i = 0; i < 64; i++) {
if ((lhs & (1ULL << i)) != 0) {
result ^= rhs << i;
}
}
return result;
};
const auto make_hi = [](uint64_t lhs, uint64_t rhs) {
uint64_t result = 0;
for (size_t i = 1; i < 64; i++) {
if ((lhs & (1ULL << i)) != 0) {
result ^= rhs >> (64 - i);
}
}
return result;
};
Dst[0] = make_lo(TMP1, TMP2);
Dst[1] = make_hi(TMP1, TMP2);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,423 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include "F80Ops.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(F80LOADFCW) {
FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle(*GetSrc<uint16_t*>(Data->SSAData, IROp->Args[0]));
}
DEF_OP(F80ADD) {
auto Op = IROp->C<IR::IROp_F80Add>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FADD(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SUB) {
auto Op = IROp->C<IR::IROp_F80Sub>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FSUB(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80MUL) {
auto Op = IROp->C<IR::IROp_F80Mul>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FMUL(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80DIV) {
auto Op = IROp->C<IR::IROp_F80Div>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FDIV(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80FYL2X) {
auto Op = IROp->C<IR::IROp_F80FYL2X>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FYL2X(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80ATAN) {
auto Op = IROp->C<IR::IROp_F80ATAN>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FATAN(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80FPREM1) {
auto Op = IROp->C<IR::IROp_F80FPREM1>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FREM1(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80FPREM) {
auto Op = IROp->C<IR::IROp_F80FPREM>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FREM(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SCALE) {
auto Op = IROp->C<IR::IROp_F80SCALE>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FSCALE(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80CVT) {
auto Op = IROp->C<IR::IROp_F80CVT>();
const uint8_t OpSize = IROp->Size;
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
switch (OpSize) {
case 4: {
float Tmp = Src;
memcpy(GDP, &Tmp, OpSize);
break;
}
case 8: {
double Tmp = Src;
memcpy(GDP, &Tmp, OpSize);
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
}
DEF_OP(F80CVTINT) {
auto Op = IROp->C<IR::IROp_F80CVTInt>();
const uint8_t OpSize = IROp->Size;
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
switch (OpSize) {
case 2: {
int16_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2)(Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
case 4: {
int32_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4)(Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
case 8: {
int64_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8)(Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
}
DEF_OP(F80CVTTO) {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch (Op->SrcSize) {
case 4: {
float Src = *GetSrc<float *>(Data->SSAData, Op->X80Src);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
case 8: {
double Src = *GetSrc<double *>(Data->SSAData, Op->X80Src);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
}
}
DEF_OP(F80CVTTOINT) {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->SrcSize) {
case 2: {
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Src);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
case 4: {
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Src);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
}
}
DEF_OP(F80ROUND) {
auto Op = IROp->C<IR::IROp_F80Round>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FRNDINT(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80F2XM1) {
auto Op = IROp->C<IR::IROp_F80F2XM1>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::F2XM1(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80TAN) {
auto Op = IROp->C<IR::IROp_F80TAN>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FTAN(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SQRT) {
auto Op = IROp->C<IR::IROp_F80SQRT>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FSQRT(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SIN) {
auto Op = IROp->C<IR::IROp_F80SIN>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FSIN(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80COS) {
auto Op = IROp->C<IR::IROp_F80COS>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FCOS(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80XTRACT_EXP) {
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FXTRACT_EXP(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80XTRACT_SIG) {
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FXTRACT_SIG(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80CMP) {
auto Op = IROp->C<IR::IROp_F80Cmp>();
uint32_t ResultFlags{};
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
bool eq, lt, nan;
X80SoftFloat::FCMP(Src1, Src2, &eq, &lt, &nan);
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
nan) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ) &&
eq) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
GD = ResultFlags;
}
DEF_OP(F80BCDLOAD) {
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->X80Src);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
// Each 4bit split is a digit
// Only 0-9 is supported, A-F results in undefined data
// | 4 bit | 4 bit |
// | 10s place | 1s place |
// EG 0x48 = 48
// EG 0x4847 = 4847
// This gives us an 18digit value encoded in BCD
// The last byte lets us know if it negative or not
for (size_t i = 0; i < 9; ++i) {
uint8_t Digit = Src1[8 - i];
// First shift our last value over
BCD *= 100;
// Add the tens place digit
BCD += (Digit >> 4) * 10;
// Add the ones place digit
BCD += Digit & 0xF;
}
// Set negative flag once converted to x87
bool Negative = Src1[9] & 0x80;
X80SoftFloat Tmp;
Tmp = BCD;
Tmp.Sign = Negative;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80BCDSTORE) {
auto Op = IROp->C<IR::IROp_F80BCDStore>();
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src));
bool Negative = Src1.Sign;
// Clear the Sign bit
Src1.Sign = 0;
uint64_t Tmp = Src1;
uint8_t BCD[10]{};
for (size_t i = 0; i < 9; ++i) {
if (Tmp == 0) {
// Nothing left? Just leave
break;
}
// Extract the lower 100 values
uint8_t Digit = Tmp % 100;
// Now divide it for the next iteration
Tmp /= 100;
uint8_t UpperNibble = Digit / 10;
uint8_t LowerNibble = Digit % 10;
// Now store the BCD
BCD[i] = (UpperNibble << 4) | LowerNibble;
}
// Set negative flag once converted to x87
BCD[9] = Negative ? 0x80 : 0;
memcpy(GDP, BCD, 10);
}
DEF_OP(F64SIN) {
auto Op = IROp->C<IR::IROp_F64SIN>();
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Tmp = sin(Src);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64COS) {
auto Op = IROp->C<IR::IROp_F64COS>();
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Tmp = cos(Src);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64TAN) {
auto Op = IROp->C<IR::IROp_F64TAN>();
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Tmp = tan(Src);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64F2XM1) {
auto Op = IROp->C<IR::IROp_F64F2XM1>();
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Tmp = exp2(Src) - 1.0;
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64ATAN) {
auto Op = IROp->C<IR::IROp_F64ATAN>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double Tmp = atan2(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64FPREM) {
auto Op = IROp->C<IR::IROp_F64FPREM>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double Tmp = fmod(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64FPREM1) {
auto Op = IROp->C<IR::IROp_F64FPREM1>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double Tmp = remainder(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64FYL2X) {
auto Op = IROp->C<IR::IROp_F64FYL2X>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double Tmp = Src2 * log2(Src1);
memcpy(GDP, &Tmp, sizeof(double));
}
DEF_OP(F64SCALE) {
auto Op = IROp->C<IR::IROp_F64SCALE>();
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
const double trunc = (double)(int64_t)(Src2); //truncate
const double Tmp = Src1 * exp2(trunc);
memcpy(GDP, &Tmp, sizeof(double));
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,399 +0,0 @@
#pragma once
#include "Common/SoftFloat.h"
#include "Common/SoftFloat-3e/softfloat.h"
#include <FEXCore/IR/IR.h>
namespace FEXCore::CPU {
template<IR::IROps Op>
struct OpHandlers {
};
template<>
struct OpHandlers<IR::OP_F80CVTTO> {
static X80SoftFloat handle4(float src) {
return src;
}
static X80SoftFloat handle8(double src) {
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80CMP> {
template<uint32_t Flags>
static uint64_t handle(X80SoftFloat Src1, X80SoftFloat Src2) {
bool eq, lt, nan;
uint64_t ResultFlags = 0;
X80SoftFloat::FCMP(Src1, Src2, &eq, &lt, &nan);
if (Flags & (1 << IR::FCMP_FLAG_LT) &&
lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
nan) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
if (Flags & (1 << IR::FCMP_FLAG_EQ) &&
eq) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
return ResultFlags;
}
};
template<>
struct OpHandlers<IR::OP_F80CVT> {
static float handle4(X80SoftFloat src) {
return src;
}
static double handle8(X80SoftFloat src) {
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80CVTINT> {
static int16_t handle2(X80SoftFloat src) {
return src;
}
static int32_t handle4(X80SoftFloat src) {
return src;
}
static int64_t handle8(X80SoftFloat src) {
return src;
}
static int16_t handle2t(X80SoftFloat src) {
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
if (rv > INT16_MAX) {
return INT16_MAX;
} else if (rv < INT16_MIN) {
return INT16_MIN;
} else {
return rv;
}
}
static int32_t handle4t(X80SoftFloat src) {
return extF80_to_i32(src, softfloat_round_minMag, false);
}
static int64_t handle8t(X80SoftFloat src) {
return extF80_to_i64(src, softfloat_round_minMag, false);
}
};
template<>
struct OpHandlers<IR::OP_F80CVTTOINT> {
static X80SoftFloat handle2(int16_t src) {
return src;
}
static X80SoftFloat handle4(int32_t src) {
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80ROUND> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FRNDINT(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80F2XM1> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::F2XM1(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80TAN> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FTAN(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80SQRT> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FSQRT(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80SIN> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FSIN(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80COS> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FCOS(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FXTRACT_EXP(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FXTRACT_SIG(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80ADD> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FADD(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80SUB> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FSUB(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80MUL> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FMUL(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80DIV> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FDIV(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FYL2X> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FYL2X(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80ATAN> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FATAN(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FPREM1> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FREM1(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FPREM> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FREM(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80SCALE> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FSCALE(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F64SIN> {
static double handle(double src) {
return sin(src);
}
};
template<>
struct OpHandlers<IR::OP_F64COS> {
static double handle(double src) {
return cos(src);
}
};
template<>
struct OpHandlers<IR::OP_F64TAN> {
static double handle(double src) {
return tan(src);
}
};
template<>
struct OpHandlers<IR::OP_F64F2XM1> {
static double handle(double src) {
return exp2(src) - 1.0;
}
};
template<>
struct OpHandlers<IR::OP_F64ATAN> {
static double handle(double src1, double src2) {
return atan2(src1, src2);
}
};
template<>
struct OpHandlers<IR::OP_F64FPREM> {
static double handle(double src1, double src2) {
return fmod(src1, src2);
}
};
template<>
struct OpHandlers<IR::OP_F64FPREM1> {
static double handle(double src1, double src2) {
return remainder(src1, src2);
}
};
template<>
struct OpHandlers<IR::OP_F64FYL2X> {
static double handle(double src1, double src2) {
return src2 * log2(src1);
}
};
template<>
struct OpHandlers<IR::OP_F64SCALE> {
static double handle(double src1, double src2) {
double trunc = (double)(int64_t)(src2); //truncate
return src1 * exp2(trunc);
}
};
template<>
struct OpHandlers<IR::OP_F80BCDSTORE> {
static X80SoftFloat handle(X80SoftFloat Src1) {
bool Negative = Src1.Sign;
Src1 = X80SoftFloat::FRNDINT(Src1);
// Clear the Sign bit
Src1.Sign = 0;
uint64_t Tmp = Src1;
X80SoftFloat Rv;
uint8_t *BCD = reinterpret_cast<uint8_t*>(&Rv);
memset(BCD, 0, 10);
for (size_t i = 0; i < 9; ++i) {
if (Tmp == 0) {
// Nothing left? Just leave
break;
}
// Extract the lower 100 values
uint8_t Digit = Tmp % 100;
// Now divide it for the next iteration
Tmp /= 100;
uint8_t UpperNibble = Digit / 10;
uint8_t LowerNibble = Digit % 10;
// Now store the BCD
BCD[i] = (UpperNibble << 4) | LowerNibble;
}
// Set negative flag once converted to x87
BCD[9] = Negative ? 0x80 : 0;
return Rv;
}
};
template<>
struct OpHandlers<IR::OP_F80BCDLOAD> {
static X80SoftFloat handle(X80SoftFloat Src) {
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
// Each 4bit split is a digit
// Only 0-9 is supported, A-F results in undefined data
// | 4 bit | 4 bit |
// | 10s place | 1s place |
// EG 0x48 = 48
// EG 0x4847 = 4847
// This gives us an 18digit value encoded in BCD
// The last byte lets us know if it negative or not
for (size_t i = 0; i < 9; ++i) {
uint8_t Digit = Src1[8 - i];
// First shift our last value over
BCD *= 100;
// Add the tens place digit
BCD += (Digit >> 4) * 10;
// Add the ones place digit
BCD += Digit & 0xF;
}
// Set negative flag once converted to x87
bool Negative = Src1[9] & 0x80;
X80SoftFloat Tmp;
Tmp = BCD;
Tmp.Sign = Negative;
return Tmp;
}
};
template<>
struct OpHandlers<IR::OP_F80LOADFCW> {
static void handle(uint16_t NewFCW) {
auto PC = (NewFCW >> 8) & 3;
switch(PC) {
case 0: extF80_roundingPrecision = 32; break;
case 2: extF80_roundingPrecision = 64; break;
case 3: extF80_roundingPrecision = 80; break;
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
}
auto RC = (NewFCW >> 10) & 3;
switch(RC) {
case 0:
softfloat_roundingMode = softfloat_round_near_even;
break;
case 1:
softfloat_roundingMode = softfloat_round_min;
break;
case 2:
softfloat_roundingMode = softfloat_round_max;
break;
case 3:
softfloat_roundingMode = softfloat_round_minMag;
break;
}
}
};
}
@@ -1,21 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(GetHostFlag) {
auto Op = IROp->C<IR::IROp_GetHostFlag>();
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Value) >> Op->Flag) & 1;
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,5 +1,6 @@
#pragma once
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
@@ -8,7 +9,6 @@
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::CPU {
class Dispatcher;
class X86DispatchGenerator;
class Arm64DispatchGenerator;
@@ -21,35 +21,32 @@ using DestMapType = std::vector<uint32_t>;
class InterpreterCore final : public CPUBackend {
public:
explicit InterpreterCore(Dispatcher *Dispatch,
FEXCore::Core::InternalThreadState *Thread);
explicit InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
~InterpreterCore() override;
std::string GetName() override { return "Interpreter"; }
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]] void *CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
bool NeedsOpDispatch() override { return true; }
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
void ClearCache() override;
bool HandleSIGBUS(int Signal, void *info, void *ucontext);
private:
size_t BufferUsed;
Dispatcher *Dispatch;
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *State;
uint32_t AllocateTmpSpace(size_t Size);
template<typename Res>
Res GetDest(void* SSAData, IR::OrderedNodeWrapper Op);
template<typename Res>
Res GetSrc(void* SSAData, IR::OrderedNodeWrapper Src);
Dispatcher *Dispatcher{};
};
template<typename T>
T AtomicCompareAndSwap(T expected, T desired, T *addr);
uint8_t AtomicFetchNeg(uint8_t *Addr);
uint16_t AtomicFetchNeg(uint16_t *Addr);
uint32_t AtomicFetchNeg(uint32_t *Addr);
uint64_t AtomicFetchNeg(uint64_t *Addr);
} // namespace FEXCore::CPU
}
@@ -1,109 +1,128 @@
#include "Common/MathUtils.h"
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/Arm64.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/DebugData.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <memory>
#include <signal.h>
#include <stdint.h>
#include <utility>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include "Interface/HLE/Thunks/Thunks.h"
#include <atomic>
#include <cmath>
#include <limits>
#include <vector>
#include "InterpreterOps.h"
#if defined(_M_X86_64)
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
#elif defined(_M_ARM_64)
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
#else
#error missing arch
#endif
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
namespace FEXCore::IR {
class IRListView;
class RegisterAllocationData;
}
namespace FEXCore::CPU {
InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::InternalThreadState *Thread)
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
, Dispatch(Dispatcher)
{
static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
auto Thread = Frame->Thread;
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
auto LocalEntry = Thread->LocalIRCache.find(Thread->CurrentFrame->State.rip);
Interpreter.FragmentExecuter = reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR);
ClearCache();
InterpreterOps::InterpretIR(Thread, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
}
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
bool InterpreterCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
#ifdef _M_ARM_64
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
}, true);
constexpr bool is_arm64 = true;
#else
constexpr bool is_arm64 = false;
#endif
}
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
const auto IRSize = AlignUp(IR->GetInlineSize(), 16);
const auto MaxSize = IRSize + Dispatcher::MaxInterpreterTrampolineSize + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
if ((BufferUsed + MaxSize) > CurrentCodeBuffer->Size) {
ThreadState->CTX->ClearCodeCache(ThreadState);
if constexpr (is_arm64) {
uint32_t *PC = reinterpret_cast<uint32_t*>(ArchHelpers::Context::GetPc(ucontext));
uint32_t Instr = PC[0];
if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
LogMan::Msg::E("Unhandled JIT SIGBUS CASPAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
return false;
}
}
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
LogMan::Msg::E("Unhandled JIT SIGBUS CASAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
return false;
}
}
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
uint8_t Op = (PC[0] >> 12) & 0xF;
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x: PC: %p Instruction: 0x%08x\n", Op, PC, PC[0]);
return false;
}
}
}
return false;
}
const auto BufferStart = CurrentCodeBuffer->Ptr + BufferUsed;
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
: CTX {ctx}
, State {Thread} {
// Grab our space for temporary data
auto DestBuffer = BufferStart;
if (!CompileThread &&
CTX->Config.Core == FEXCore::Config::CONFIG_INTERPRETER) {
CreateAsmDispatch(ctx, Thread);
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
});
if (GDBEnabled) {
const auto GDBSize = Dispatch->GenerateGDBPauseCheck(DestBuffer, Entry);
DestBuffer += GDBSize;
BufferUsed += GDBSize;
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->HandleSIGBUS(Signal, info, ucontext);
});
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
};
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
}
}
const auto TrampolineSize = Dispatch->GenerateInterpreterTrampoline(DestBuffer);
DestBuffer += TrampolineSize;
BufferUsed += TrampolineSize;
IR->Serialize(DestBuffer);
DestBuffer += IRSize;
BufferUsed += IRSize;
return BufferStart;
}
void InterpreterCore::ClearCache() {
// Calling this one is needed to setup the initial CurrentCodeBuffer
[[maybe_unused]] auto CodeBuffer = GetEmptyCodeBuffer();
BufferUsed = 0;
InterpreterCore::~InterpreterCore() {
delete Dispatcher;
}
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
return std::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
void *InterpreterCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
return reinterpret_cast<void*>(InterpreterExecution);
}
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
InterpreterCore::InitializeSignalHandlers(CTX);
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
return new InterpreterCore(ctx, Thread, CompileThread);
}
CPUBackendFeatures GetInterpreterBackendFeatures() {
return CPUBackendFeatures { };
}
}
@@ -1,7 +1,5 @@
#pragma once
#include <memory>
namespace FEXCore::Context {
struct Context;
}
@@ -12,11 +10,7 @@ namespace FEXCore::Core {
namespace FEXCore::CPU {
class CPUBackend;
struct DispatcherConfig;
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread);
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
CPUBackendFeatures GetInterpreterBackendFeatures();
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
} // namespace FEXCore::CPU
}
@@ -1,184 +0,0 @@
#pragma once
#include <FEXCore/IR/IR.h>
#define GD *GetDest<uint64_t*>(Data->SSAData, Node)
#define GDP GetDest<void*>(Data->SSAData, Node)
#define DO_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(GDP); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
*Dst_d = func(*Src1_d, *Src2_d); \
break; \
}
#define DO_SCALAR_COMPARE_OP(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
Dst_d[0] = func(Src1_d[0], Src2_d[0]); \
break; \
}
#define DO_VECTOR_COMPARE_OP(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
} \
break; \
}
#define DO_VECTOR_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
} \
break; \
}
#define DO_VECTOR_PAIR_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i*2], Src1_d[i*2 + 1]); \
Dst_d[i+Elements] = func(Src2_d[i*2], Src2_d[i*2 + 1]); \
} \
break; \
}
#define DO_VECTOR_SCALAR_OP(size, type, func)\
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], *Src2_d); \
} \
break; \
}
#define DO_VECTOR_0SRC_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(); \
} \
break; \
}
#define DO_VECTOR_1SRC_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src_d[i]); \
} \
break; \
}
#define DO_VECTOR_REDUCE_1SRC_OP(size, type, func, start_val) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type*>(Src); \
type begin = start_val; \
for (uint8_t i = 0; i < Elements; ++i) { \
begin = func(begin, Src_d[i]); \
} \
Dst_d[0] = begin; \
break; \
}
#define DO_VECTOR_SAT_OP(size, type, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], Src2_d[i], min, max); \
} \
break; \
}
#define DO_VECTOR_1SRC_2TYPE_OP(size, type, type2, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func(Src_d[i], min, max); \
} \
break; \
}
#define DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(type, type2, func, min, max) \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func(Src_d[i], min, max); \
}
#define DO_VECTOR_1SRC_2TYPE_OP_TOP(size, type, type2, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src2); \
memcpy(Dst_d, Src1, Elements * sizeof(type2));\
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i+Elements] = (type)func(Src_d[i], min, max); \
} \
break; \
}
#define DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(size, type, type2, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func(Src_d[i+Elements], min, max); \
} \
break; \
}
#define DO_VECTOR_2SRC_2TYPE_OP(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func((type)Src1_d[i], (type)Src2_d[i]); \
} \
break; \
}
#define DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func((type)Src1_d[i+Elements], (type)Src2_d[i+Elements]); \
} \
break; \
}
struct InterpVector256 {
__uint128_t Lower;
__uint128_t Upper;
};
template<typename Res>
Res GetDest(void* SSAData, FEXCore::IR::OrderedNodeWrapper Op) {
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.ID().Value];
return reinterpret_cast<Res>(DstPtr);
}
template<typename Res>
Res GetDest(void* SSAData, FEXCore::IR::NodeID Op) {
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.Value];
return reinterpret_cast<Res>(DstPtr);
}
template<typename Res>
Res GetSrc(void* SSAData, FEXCore::IR::OrderedNodeWrapper Src) {
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Src.ID().Value];
return reinterpret_cast<Res>(DstPtr);
}
@@ -1,313 +0,0 @@
#include "FEXCore/Core/CoreState.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/F80Ops.h"
#include <cstddef>
#include <cstdint>
namespace FEXCore::CPU {
template<typename R, typename... Args>
static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_UNKNOWN, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F32, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_VOID_U16, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I32, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F32_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F64_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I16_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I32_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_F80_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F80_F80, (void*)fn, HandlerIndex};
}
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
Info[Core::OPINDEX_F80LOADFCW] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW).fn);
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4).fn);
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8).fn);
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4).fn);
Info[Core::OPINDEX_F80CVT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8).fn);
Info[Core::OPINDEX_F80CVTINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2).fn);
Info[Core::OPINDEX_F80CVTINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4).fn);
Info[Core::OPINDEX_F80CVTINT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8).fn);
Info[Core::OPINDEX_F80CVTINT_TRUNC2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2).fn);
Info[Core::OPINDEX_F80CVTINT_TRUNC4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4).fn);
Info[Core::OPINDEX_F80CVTINT_TRUNC8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8).fn);
Info[Core::OPINDEX_F80CMP_0] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>, Core::OPINDEX_F80CMP_0).fn);
Info[Core::OPINDEX_F80CMP_1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>, Core::OPINDEX_F80CMP_1).fn);
Info[Core::OPINDEX_F80CMP_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>, Core::OPINDEX_F80CMP_2).fn);
Info[Core::OPINDEX_F80CMP_3] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>, Core::OPINDEX_F80CMP_3).fn);
Info[Core::OPINDEX_F80CMP_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>, Core::OPINDEX_F80CMP_4).fn);
Info[Core::OPINDEX_F80CMP_5] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>, Core::OPINDEX_F80CMP_5).fn);
Info[Core::OPINDEX_F80CMP_6] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>, Core::OPINDEX_F80CMP_6).fn);
Info[Core::OPINDEX_F80CMP_7] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>, Core::OPINDEX_F80CMP_7).fn);
Info[Core::OPINDEX_F80CVTTOINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2).fn);
Info[Core::OPINDEX_F80CVTTOINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4).fn);
// Unary
Info[Core::OPINDEX_F80ROUND] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ROUND>::handle, Core::OPINDEX_F80ROUND).fn);
Info[Core::OPINDEX_F80F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80F2XM1>::handle, Core::OPINDEX_F80F2XM1).fn);
Info[Core::OPINDEX_F80TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80TAN>::handle, Core::OPINDEX_F80TAN).fn);
Info[Core::OPINDEX_F80SQRT] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SQRT>::handle, Core::OPINDEX_F80SQRT).fn);
Info[Core::OPINDEX_F80SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SIN>::handle, Core::OPINDEX_F80SIN).fn);
Info[Core::OPINDEX_F80COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80COS>::handle, Core::OPINDEX_F80COS).fn);
Info[Core::OPINDEX_F80XTRACT_EXP] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle, Core::OPINDEX_F80XTRACT_EXP).fn);
Info[Core::OPINDEX_F80XTRACT_SIG] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle, Core::OPINDEX_F80XTRACT_SIG).fn);
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle, Core::OPINDEX_F80BCDSTORE).fn);
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle, Core::OPINDEX_F80BCDLOAD).fn);
// Binary
Info[Core::OPINDEX_F80ADD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ADD>::handle, Core::OPINDEX_F80ADD).fn);
Info[Core::OPINDEX_F80SUB] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SUB>::handle, Core::OPINDEX_F80SUB).fn);
Info[Core::OPINDEX_F80MUL] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80MUL>::handle, Core::OPINDEX_F80MUL).fn);
Info[Core::OPINDEX_F80DIV] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80DIV>::handle, Core::OPINDEX_F80DIV).fn);
Info[Core::OPINDEX_F80FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2X>::handle, Core::OPINDEX_F80FYL2X).fn);
Info[Core::OPINDEX_F80ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ATAN>::handle, Core::OPINDEX_F80ATAN).fn);
Info[Core::OPINDEX_F80FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM1>::handle, Core::OPINDEX_F80FPREM1).fn);
Info[Core::OPINDEX_F80FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM>::handle, Core::OPINDEX_F80FPREM).fn);
Info[Core::OPINDEX_F80SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle, Core::OPINDEX_F80SCALE).fn);
// Double Precision
Info[Core::OPINDEX_F64SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle, Core::OPINDEX_F64SIN).fn);
Info[Core::OPINDEX_F64COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle, Core::OPINDEX_F64COS).fn);
Info[Core::OPINDEX_F64TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle, Core::OPINDEX_F64TAN).fn);
Info[Core::OPINDEX_F64ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle, Core::OPINDEX_F64ATAN).fn);
Info[Core::OPINDEX_F64F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle, Core::OPINDEX_F64F2XM1).fn);
Info[Core::OPINDEX_F64FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle, Core::OPINDEX_F64FYL2X).fn);
Info[Core::OPINDEX_F64FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle, Core::OPINDEX_F64FPREM).fn);
Info[Core::OPINDEX_F64FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle, Core::OPINDEX_F64FPREM1).fn);
Info[Core::OPINDEX_F64SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle, Core::OPINDEX_F64SCALE).fn);
}
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
uint8_t OpSize = IROp->Size;
switch(IROp->Op) {
case IR::OP_F80LOADFCW: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW);
return true;
}
case IR::OP_F80CVTTO: {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch (Op->SrcSize) {
case 4: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4);
return true;
}
case 8: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8);
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CVT: {
switch (OpSize) {
case 4: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4);
return true;
}
case 8: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8);
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CVTINT: {
auto Op = IROp->C<IR::IROp_F80CVTInt>();
switch (OpSize) {
case 2: {
if (Op->Truncate) {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2);
}
else {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2);
}
return true;
}
case 4: {
if (Op->Truncate) {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4);
}
else {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4);
}
return true;
}
case 8: {
if (Op->Truncate) {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8);
}
else {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8);
}
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CMP: {
auto Op = IROp->C<IR::IROp_F80Cmp>();
static constexpr std::array handlers{
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
};
*Info = GetFallbackInfo(handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags));
return true;
}
case IR::OP_F80CVTTOINT: {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->SrcSize) {
case 2: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2);
return true;
}
case 4: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4);
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
#define COMMON_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP); \
return true; \
}
#define COMMON_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
return true; \
}
// Unary
COMMON_X87_OP(ROUND)
COMMON_X87_OP(F2XM1)
COMMON_X87_OP(TAN)
COMMON_X87_OP(SQRT)
COMMON_X87_OP(SIN)
COMMON_X87_OP(COS)
COMMON_X87_OP(XTRACT_EXP)
COMMON_X87_OP(XTRACT_SIG)
COMMON_X87_OP(BCDSTORE)
COMMON_X87_OP(BCDLOAD)
// Binary
COMMON_X87_OP(ADD)
COMMON_X87_OP(SUB)
COMMON_X87_OP(MUL)
COMMON_X87_OP(DIV)
COMMON_X87_OP(FYL2X)
COMMON_X87_OP(ATAN)
COMMON_X87_OP(FPREM1)
COMMON_X87_OP(FPREM)
COMMON_X87_OP(SCALE)
// Double Precision Unary
COMMON_F64_OP(F2XM1)
COMMON_F64_OP(TAN)
COMMON_F64_OP(SIN)
COMMON_F64_OP(COS)
// Double Precision Binary
COMMON_F64_OP(FYL2X)
COMMON_F64_OP(ATAN)
COMMON_F64_OP(FPREM1)
COMMON_F64_OP(FPREM)
COMMON_F64_OP(SCALE)
default:
break;
}
return false;
}
}
File diff suppressed because it is too large. Load diff
@@ -1,17 +1,9 @@
#pragma once
#include <stdint.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::Core {
struct InternalThreadState;
}
namespace FEXCore::IR {
class IRListView;
struct IROp_Header;
}
namespace FEXCore::Core{
@@ -28,8 +20,6 @@ namespace FEXCore::CPU {
FABI_F80_I32,
FABI_F32_F80,
FABI_F64_F80,
FABI_F64_F64,
FABI_F64_F64_F64,
FABI_I16_F80,
FABI_I32_F80,
FABI_I64_F80,
@@ -41,369 +31,12 @@ namespace FEXCore::CPU {
struct FallbackInfo {
FallbackABI ABI;
void *fn;
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
};
class InterpreterOps {
public:
static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR);
static void FillFallbackIndexPointers(uint64_t *Info);
static bool GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info);
struct IROpData {
FEXCore::Core::InternalThreadState *State{};
uint64_t CurrentEntry{};
FEXCore::IR::IRListView const *CurrentIR{};
volatile void *StackEntry{};
void *SSAData{};
struct {
bool Quit;
bool Redo;
} BlockResults{};
IR::NodeIterator BlockIterator{0, 0};
};
#define DEF_OP(x) static void Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
///< Unhandled handler
DEF_OP(Unhandled);
///< No-op Handler
DEF_OP(NoOp);
///< ALU Ops
DEF_OP(TruncElementPair);
DEF_OP(Constant);
DEF_OP(EntrypointOffset);
DEF_OP(InlineConstant);
DEF_OP(InlineEntrypointOffset);
DEF_OP(CycleCounter);
DEF_OP(Add);
DEF_OP(Sub);
DEF_OP(Neg);
DEF_OP(Mul);
DEF_OP(UMul);
DEF_OP(Div);
DEF_OP(UDiv);
DEF_OP(Rem);
DEF_OP(URem);
DEF_OP(MulH);
DEF_OP(UMulH);
DEF_OP(Or);
DEF_OP(And);
DEF_OP(Andn);
DEF_OP(Xor);
DEF_OP(Lshl);
DEF_OP(Lshr);
DEF_OP(Ashr);
DEF_OP(Rol);
DEF_OP(Ror);
DEF_OP(Extr);
DEF_OP(PDep);
DEF_OP(PExt);
DEF_OP(LDiv);
DEF_OP(LUDiv);
DEF_OP(LRem);
DEF_OP(LURem);
DEF_OP(Zext);
DEF_OP(Not);
DEF_OP(Popcount);
DEF_OP(FindLSB);
DEF_OP(FindMSB);
DEF_OP(FindTrailingZeros);
DEF_OP(CountLeadingZeroes);
DEF_OP(Rev);
DEF_OP(Bfi);
DEF_OP(Bfe);
DEF_OP(Sbfe);
DEF_OP(Select);
DEF_OP(VExtractToGPR);
DEF_OP(Float_ToGPR_ZU);
DEF_OP(Float_ToGPR_ZS);
DEF_OP(Float_ToGPR_S);
DEF_OP(FCmp);
///< Atomic ops
DEF_OP(CASPair);
DEF_OP(CAS);
DEF_OP(AtomicAdd);
DEF_OP(AtomicSub);
DEF_OP(AtomicAnd);
DEF_OP(AtomicOr);
DEF_OP(AtomicXor);
DEF_OP(AtomicSwap);
DEF_OP(AtomicFetchAdd);
DEF_OP(AtomicFetchSub);
DEF_OP(AtomicFetchAnd);
DEF_OP(AtomicFetchOr);
DEF_OP(AtomicFetchXor);
DEF_OP(AtomicFetchNeg);
///< Branch ops
DEF_OP(SignalReturn);
DEF_OP(CallbackReturn);
DEF_OP(ExitFunction);
DEF_OP(Jump);
DEF_OP(CondJump);
DEF_OP(Syscall);
DEF_OP(InlineSyscall);
DEF_OP(Thunk);
DEF_OP(ValidateCode);
DEF_OP(ThreadRemoveCodeEntry);
DEF_OP(CPUID);
///< Conversion ops
DEF_OP(VInsGPR);
DEF_OP(VCastFromGPR);
DEF_OP(Float_FromGPR_S);
DEF_OP(Float_FToF);
DEF_OP(Vector_SToF);
DEF_OP(Vector_FToZS);
DEF_OP(Vector_FToS);
DEF_OP(Vector_FToF);
DEF_OP(Vector_FToI);
///< Flag ops
DEF_OP(GetHostFlag);
///< Memory ops
DEF_OP(LoadContext);
DEF_OP(StoreContext);
DEF_OP(LoadRegister);
DEF_OP(StoreRegister);
DEF_OP(LoadContextIndexed);
DEF_OP(StoreContextIndexed);
DEF_OP(SpillRegister);
DEF_OP(FillRegister);
DEF_OP(LoadFlag);
DEF_OP(StoreFlag);
DEF_OP(LoadMem);
DEF_OP(StoreMem);
DEF_OP(CacheLineClear);
DEF_OP(CacheLineClean);
DEF_OP(CacheLineZero);
///< Misc ops
DEF_OP(EndBlock);
DEF_OP(Fence);
DEF_OP(Break);
DEF_OP(Phi);
DEF_OP(PhiValue);
DEF_OP(Print);
DEF_OP(GetRoundingMode);
DEF_OP(SetRoundingMode);
DEF_OP(ProcessorID);
DEF_OP(RDRAND);
DEF_OP(Yield);
///< Move ops
DEF_OP(ExtractElementPair);
DEF_OP(CreateElementPair);
DEF_OP(Mov);
///< Vector ops
DEF_OP(VectorZero);
DEF_OP(VectorImm);
DEF_OP(VMov);
DEF_OP(VAnd);
DEF_OP(VBic);
DEF_OP(VOr);
DEF_OP(VXor);
DEF_OP(VAdd);
DEF_OP(VSub);
DEF_OP(VUQAdd);
DEF_OP(VUQSub);
DEF_OP(VSQAdd);
DEF_OP(VSQSub);
DEF_OP(VAddP);
DEF_OP(VAddV);
DEF_OP(VUMinV);
DEF_OP(VURAvg);
DEF_OP(VAbs);
DEF_OP(VPopcount);
DEF_OP(VFAdd);
DEF_OP(VFAddP);
DEF_OP(VFSub);
DEF_OP(VFMul);
DEF_OP(VFDiv);
DEF_OP(VFMin);
DEF_OP(VFMax);
DEF_OP(VFRecp);
DEF_OP(VFSqrt);
DEF_OP(VFRSqrt);
DEF_OP(VNeg);
DEF_OP(VFNeg);
DEF_OP(VNot);
DEF_OP(VUMin);
DEF_OP(VSMin);
DEF_OP(VUMax);
DEF_OP(VSMax);
DEF_OP(VZip);
DEF_OP(VUnZip);
DEF_OP(VBSL);
DEF_OP(VCMPEQ);
DEF_OP(VCMPEQZ);
DEF_OP(VCMPGT);
DEF_OP(VCMPGTZ);
DEF_OP(VCMPLTZ);
DEF_OP(VFCMPEQ);
DEF_OP(VFCMPNEQ);
DEF_OP(VFCMPLT);
DEF_OP(VFCMPGT);
DEF_OP(VFCMPLE);
DEF_OP(VFCMPORD);
DEF_OP(VFCMPUNO);
DEF_OP(VUShl);
DEF_OP(VUShr);
DEF_OP(VSShr);
DEF_OP(VUShlS);
DEF_OP(VUShrS);
DEF_OP(VSShrS);
DEF_OP(VInsElement);
DEF_OP(VDupElement);
DEF_OP(VExtr);
DEF_OP(VUShrI);
DEF_OP(VSShrI);
DEF_OP(VShlI);
DEF_OP(VUShrNI);
DEF_OP(VUShrNI2);
DEF_OP(VSXTL);
DEF_OP(VSXTL2);
DEF_OP(VUXTL);
DEF_OP(VUXTL2);
DEF_OP(VSQXTN);
DEF_OP(VSQXTN2);
DEF_OP(VSQXTUN);
DEF_OP(VSQXTUN2);
DEF_OP(VUMul);
DEF_OP(VUMull);
DEF_OP(VSMul);
DEF_OP(VSMull);
DEF_OP(VUMull2);
DEF_OP(VSMull2);
DEF_OP(VUABDL);
DEF_OP(VTBL1);
DEF_OP(VRev64);
///< Encryption ops
DEF_OP(AESImc);
DEF_OP(AESEnc);
DEF_OP(AESEncLast);
DEF_OP(AESDec);
DEF_OP(AESDecLast);
DEF_OP(AESKeyGenAssist);
DEF_OP(CRC32);
DEF_OP(PCLMUL);
///< F80 ops
DEF_OP(F80LOADFCW);
DEF_OP(F80ADD);
DEF_OP(F80SUB);
DEF_OP(F80MUL);
DEF_OP(F80DIV);
DEF_OP(F80FYL2X);
DEF_OP(F80ATAN);
DEF_OP(F80FPREM1);
DEF_OP(F80FPREM);
DEF_OP(F80SCALE);
DEF_OP(F80CVT);
DEF_OP(F80CVTINT);
DEF_OP(F80CVTTO);
DEF_OP(F80CVTTOINT);
DEF_OP(F80ROUND);
DEF_OP(F80F2XM1);
DEF_OP(F80TAN);
DEF_OP(F80SQRT);
DEF_OP(F80SIN);
DEF_OP(F80COS);
DEF_OP(F80XTRACT_EXP);
DEF_OP(F80XTRACT_SIG);
DEF_OP(F80CMP);
DEF_OP(F80BCDLOAD);
DEF_OP(F80BCDSTORE);
//< F64 ops
DEF_OP(F64SIN);
DEF_OP(F64COS);
DEF_OP(F64TAN);
DEF_OP(F64F2XM1);
DEF_OP(F64ATAN);
DEF_OP(F64FPREM);
DEF_OP(F64FPREM1);
DEF_OP(F64FYL2X);
DEF_OP(F64SCALE);
#undef DEF_OP
template<typename unsigned_type, typename signed_type, typename float_type>
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
bool CompResult = false;
switch (Cond) {
case FEXCore::IR::COND_EQ:
CompResult = static_cast<unsigned_type>(Src1) == static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_NEQ:
CompResult = static_cast<unsigned_type>(Src1) != static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_SGE:
CompResult = static_cast<signed_type>(Src1) >= static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_SLT:
CompResult = static_cast<signed_type>(Src1) < static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_SGT:
CompResult = static_cast<signed_type>(Src1) > static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_SLE:
CompResult = static_cast<signed_type>(Src1) <= static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_UGE:
CompResult = static_cast<unsigned_type>(Src1) >= static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_ULT:
CompResult = static_cast<unsigned_type>(Src1) < static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_UGT:
CompResult = static_cast<unsigned_type>(Src1) > static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_ULE:
CompResult = static_cast<unsigned_type>(Src1) <= static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_FLU:
CompResult = reinterpret_cast<float_type&>(Src1) < reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FGE:
CompResult = reinterpret_cast<float_type&>(Src1) >= reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FLEU:
CompResult = reinterpret_cast<float_type&>(Src1) <= reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FGT:
CompResult = reinterpret_cast<float_type&>(Src1) > reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FU:
CompResult = (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FNU:
CompResult = !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_MI:
case FEXCore::IR::COND_PL:
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
break;
}
return CompResult;
}
static uint8_t GetOpSize(FEXCore::IR::IRListView const *CurrentIR, IR::OrderedNodeWrapper Node) {
auto IROp = CurrentIR->GetOp<FEXCore::IR::IROp_Header>(Node);
return IROp->Size;
}
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
};
} // namespace FEXCore::CPU
};
@@ -1,327 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
static inline void CacheLineFlush(char *Addr) {
#ifdef _M_X86_64
__asm volatile (
"clflush (%[Addr]);"
:: [Addr] "r" (Addr)
: "memory");
#else
__builtin___clear_cache(Addr, Addr+64);
#endif
}
static inline void CacheLineClean(char *Addr) {
#ifdef _M_X86_64
__asm volatile (
"clwb (%[Addr]);"
:: [Addr] "r" (Addr)
: "memory");
#elif _M_ARM_64
__asm volatile (
"dc cvac, %[Addr]"
:: [Addr] "r" (Addr)
: "memory");
#else
LOGMAN_THROW_A_FMT("Unsupported architecture with cacheline clean");
#endif
}
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(LoadContext) {
const auto Op = IROp->C<IR::IROp_LoadContext>();
const auto OpSize = IROp->Size;
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Src = ContextPtr + Op->Offset;
#define LOAD_CTX(x, y) \
case x: { \
y const *MemData = reinterpret_cast<y const*>(Src); \
GD = *MemData; \
break; \
}
switch (OpSize) {
LOAD_CTX(1, uint8_t)
LOAD_CTX(2, uint16_t)
LOAD_CTX(4, uint32_t)
LOAD_CTX(8, uint64_t)
case 16:
case 32: {
void const *MemData = reinterpret_cast<void const*>(Src);
memcpy(GDP, MemData, OpSize);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
break;
}
#undef LOAD_CTX
}
DEF_OP(StoreContext) {
const auto Op = IROp->C<IR::IROp_StoreContext>();
const auto OpSize = IROp->Size;
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Dst = ContextPtr + Op->Offset;
void *MemData = reinterpret_cast<void*>(Dst);
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
memcpy(MemData, Src, OpSize);
}
DEF_OP(LoadRegister) {
const auto Op = IROp->C<IR::IROp_LoadRegister>();
const auto OpSize = IROp->Size;
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Src = ContextPtr + Op->Offset;
#define LOAD_CTX(x, y) \
case x: { \
y const *MemData = reinterpret_cast<y const*>(Src); \
GD = *MemData; \
break; \
}
switch (OpSize) {
LOAD_CTX(1, uint8_t)
LOAD_CTX(2, uint16_t)
LOAD_CTX(4, uint32_t)
LOAD_CTX(8, uint64_t)
case 16:
case 32: {
void const *MemData = reinterpret_cast<void const*>(Src);
memcpy(GDP, MemData, OpSize);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
break;
}
#undef LOAD_CTX
}
DEF_OP(StoreRegister) {
const auto Op = IROp->C<IR::IROp_StoreRegister>();
const auto OpSize = IROp->Size;
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Dst = ContextPtr + Op->Offset;
void *MemData = reinterpret_cast<void*>(Dst);
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
memcpy(MemData, Src, OpSize);
}
DEF_OP(LoadContextIndexed) {
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
const auto OpSize = IROp->Size;
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Src = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
#define LOAD_CTX(x, y) \
case x: { \
y const *MemData = reinterpret_cast<y const*>(Src); \
GD = *MemData; \
break; \
}
switch (OpSize) {
LOAD_CTX(1, uint8_t)
LOAD_CTX(2, uint16_t)
LOAD_CTX(4, uint32_t)
LOAD_CTX(8, uint64_t)
case 16:
case 32: {
void const *MemData = reinterpret_cast<void const*>(Src);
memcpy(GDP, MemData, OpSize);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
break;
}
#undef LOAD_CTX
}
DEF_OP(StoreContextIndexed) {
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
const auto OpSize = IROp->Size;
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Dst = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
void *MemData = reinterpret_cast<void*>(Dst);
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
memcpy(MemData, Src, OpSize);
}
DEF_OP(SpillRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(FillRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(LoadFlag) {
auto Op = IROp->C<IR::IROp_LoadFlag>();
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
ContextPtr += Op->Flag;
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
GD = *MemData;
}
DEF_OP(StoreFlag) {
auto Op = IROp->C<IR::IROp_StoreFlag>();
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
ContextPtr += Op->Flag;
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
*MemData = Arg;
}
DEF_OP(LoadMem) {
const auto Op = IROp->C<IR::IROp_LoadMem>();
const auto OpSize = IROp->Size;
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
if (!Op->Offset.IsInvalid()) {
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
switch(Op->OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
}
}
memset(GDP, 0, Core::CPUState::XMM_AVX_REG_SIZE);
switch (OpSize) {
case 1: {
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
GD = D->load();
break;
}
case 2: {
auto D = reinterpret_cast<const std::atomic<uint16_t>*>(MemData);
GD = D->load();
break;
}
case 4: {
auto D = reinterpret_cast<const std::atomic<uint32_t>*>(MemData);
GD = D->load();
break;
}
case 8: {
auto D = reinterpret_cast<const std::atomic<uint64_t>*>(MemData);
GD = D->load();
break;
}
default:
memcpy(GDP, MemData, OpSize);
break;
}
}
DEF_OP(StoreMem) {
const auto Op = IROp->C<IR::IROp_StoreMem>();
const auto OpSize = IROp->Size;
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
if (!Op->Offset.IsInvalid()) {
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
switch(Op->OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
}
}
switch (OpSize) {
case 1: {
reinterpret_cast<std::atomic<uint8_t>*>(MemData)->store(*GetSrc<uint8_t*>(Data->SSAData, Op->Value));
break;
}
case 2: {
reinterpret_cast<std::atomic<uint16_t>*>(MemData)->store(*GetSrc<uint16_t*>(Data->SSAData, Op->Value));
break;
}
case 4: {
reinterpret_cast<std::atomic<uint32_t>*>(MemData)->store(*GetSrc<uint32_t*>(Data->SSAData, Op->Value));
break;
}
case 8: {
reinterpret_cast<std::atomic<uint64_t>*>(MemData)->store(*GetSrc<uint64_t*>(Data->SSAData, Op->Value));
break;
}
default:
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
break;
}
}
DEF_OP(CacheLineClear) {
auto Op = IROp->C<IR::IROp_CacheLineClear>();
char *MemData = *GetSrc<char **>(Data->SSAData, Op->Addr);
// 64-byte cache line clear
CacheLineFlush(MemData);
}
DEF_OP(CacheLineClean) {
auto Op = IROp->C<IR::IROp_CacheLineClean>();
char *MemData = *GetSrc<char **>(Data->SSAData, Op->Addr);
// 64-byte cache line clear
CacheLineClean(MemData);
}
DEF_OP(CacheLineZero) {
auto Op = IROp->C<IR::IROp_CacheLineZero>();
uintptr_t MemData = *GetSrc<uintptr_t*>(Data->SSAData, Op->Addr);
// Force cacheline alignment
MemData = MemData & ~(CPUIDEmu::CACHELINE_SIZE - 1);
using DataType = uint64_t;
DataType *MemData64 = reinterpret_cast<DataType*>(MemData);
// 64-byte cache line zero
for (size_t i = 0; i < (CPUIDEmu::CACHELINE_SIZE / sizeof(DataType)); ++i) {
MemData64[i] = 0;
}
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,174 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <FEXHeaderUtils/Syscalls.h>
#include <cstdint>
#ifdef _M_X86_64
#include <xmmintrin.h>
#endif
#include <sys/random.h>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(Fence) {
auto Op = IROp->C<IR::IROp_Fence>();
switch (Op->Fence) {
case IR::Fence_Load.Val:
std::atomic_thread_fence(std::memory_order_acquire);
break;
case IR::Fence_LoadStore.Val:
std::atomic_thread_fence(std::memory_order_seq_cst);
break;
case IR::Fence_Store.Val:
std::atomic_thread_fence(std::memory_order_release);
break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
}
}
DEF_OP(Break) {
auto Op = IROp->C<IR::IROp_Break>();
Data->State->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = 1;
Data->State->CurrentFrame->SynchronousFaultData.Signal = Op->Reason.Signal;
Data->State->CurrentFrame->SynchronousFaultData.TrapNo = Op->Reason.TrapNumber;
Data->State->CurrentFrame->SynchronousFaultData.err_code = Op->Reason.ErrorRegister;
Data->State->CurrentFrame->SynchronousFaultData.si_code = Op->Reason.si_code;
switch (Op->Reason.Signal) {
case SIGILL:
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
break;
case SIGTRAP:
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
break;
case SIGSEGV:
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGSEGV);
break;
default:
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
break;
}
}
DEF_OP(GetRoundingMode) {
uint32_t GuestRounding{};
#ifdef _M_ARM_64
uint64_t Tmp{};
__asm(R"(
mrs %[Tmp], FPCR;
)"
: [Tmp] "=r" (Tmp));
// Extract the rounding
// On ARM the ordering is different than on x86
GuestRounding |= ((Tmp >> 24) & 1) ? IR::ROUND_MODE_FLUSH_TO_ZERO : 0;
uint8_t RoundingMode = (Tmp >> 22) & 0b11;
if (RoundingMode == 0)
GuestRounding |= IR::ROUND_MODE_NEAREST;
else if (RoundingMode == 1)
GuestRounding |= IR::ROUND_MODE_POSITIVE_INFINITY;
else if (RoundingMode == 2)
GuestRounding |= IR::ROUND_MODE_NEGATIVE_INFINITY;
else if (RoundingMode == 3)
GuestRounding |= IR::ROUND_MODE_TOWARDS_ZERO;
#else
GuestRounding = _mm_getcsr();
// Extract the rounding
GuestRounding = (GuestRounding >> 13) & 0b111;
#endif
memcpy(GDP, &GuestRounding, sizeof(GuestRounding));
}
DEF_OP(SetRoundingMode) {
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
const auto GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->RoundMode);
#ifdef _M_ARM_64
uint64_t HostRounding{};
__asm volatile(R"(
mrs %[Tmp], FPCR;
)"
: [Tmp] "=r" (HostRounding));
// Mask out the rounding
HostRounding &= ~(0b111 << 22);
HostRounding |= (GuestRounding & IR::ROUND_MODE_FLUSH_TO_ZERO) ? (1U << 24) : 0;
uint8_t RoundingMode = GuestRounding & 0b11;
if (RoundingMode == IR::ROUND_MODE_NEAREST)
HostRounding |= (0b00U << 22);
else if (RoundingMode == IR::ROUND_MODE_POSITIVE_INFINITY)
HostRounding |= (0b01U << 22);
else if (RoundingMode == IR::ROUND_MODE_NEGATIVE_INFINITY)
HostRounding |= (0b10U << 22);
else if (RoundingMode == IR::ROUND_MODE_TOWARDS_ZERO)
HostRounding |= (0b11U << 22);
__asm volatile(R"(
msr FPCR, %[Tmp];
)"
:: [Tmp] "r" (HostRounding));
#else
uint32_t HostRounding = _mm_getcsr();
// Cut out the host rounding mode
HostRounding &= ~(0b111 << 13);
// Insert our new rounding mode
HostRounding |= GuestRounding << 13;
_mm_setcsr(HostRounding);
#endif
}
DEF_OP(Print) {
auto Op = IROp->C<IR::IROp_Print>();
const uint8_t OpSize = IROp->Size;
if (OpSize <= 8) {
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
LogMan::Msg::IFmt(">>>> Value in Arg: 0x{:x}, {}", Src, Src);
}
else if (OpSize == 16) {
const auto Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Value);
const uint64_t Src0 = Src;
const uint64_t Src1 = Src >> 64;
LogMan::Msg::IFmt(">>>> Value[0] in Arg: 0x{:x}, {}", Src0, Src0);
LogMan::Msg::IFmt(" Value[1] in Arg: 0x{:x}, {}", Src1, Src1);
}
else
LOGMAN_MSG_A_FMT("Unknown value size: {}", OpSize);
}
DEF_OP(ProcessorID) {
uint32_t CPU, CPUNode;
FHU::Syscalls::getcpu(&CPU, &CPUNode);
GD = (CPUNode << 12) | CPU;
}
DEF_OP(RDRAND) {
// We are ignoring Op->GetReseeded in the interpreter
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
ssize_t Result = ::getrandom(&DstPtr[0], 8, 0);
// Second result is if we managed to read a valid random number or not
DstPtr[1] = Result == 8 ? 1 : 0;
}
DEF_OP(Yield) {
// Nop implementation
}
#undef DEF_OP
} // namespace FEXCore::CPU
Loaded 100 of 1296 files, more files were not shown because too many files have changed in this diff. Show more