mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-08 21:00:19 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
33fe6813fc |
No files matched your search
@@ -1,45 +0,0 @@
|
||||
---
|
||||
name: Potential Game Bug
|
||||
about: A bug in FEX-Emu that causes a problem in a game
|
||||
title: "[Game]: [Short Problem Description]"
|
||||
labels: Game related
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**What Game**
|
||||
The game name.
|
||||
A link to the storefront where to get the game. GOG, Steam, Itch.io, etc
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
**To Reproduce**
|
||||
Steps to reproduce the behavior:
|
||||
1. Go to '...'
|
||||
2. Click on '....'
|
||||
3. Scroll down to '....'
|
||||
4. See error
|
||||
|
||||
**Expected behavior**
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
**Screenshots and Video**
|
||||
If applicable, add screenshots and video to help explain your problem.
|
||||
|
||||
**System information:**
|
||||
- OS: [eg: Ubuntu 21.10]
|
||||
- CPU/SoC: [eg: Snapdragon 888, Intel Core i8-12900k]
|
||||
- Video driver version: [eg: OpenGL ES 3.2 Mesa 22.0.0-devel (git-9ff086052a)]
|
||||
- RootFS used: [eg: Ubuntu 21.10 Official Rootfs]
|
||||
- FEX version: (FEXGetConfig --version) [eg: FEX-2112-155-gc691d709]
|
||||
- Thunks Enabled: [Yes/No]
|
||||
|
||||
**Additional context**
|
||||
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
|
||||
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
|
||||
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
|
||||
- Is this a Vulkan game: [Yes/No/Unknown]
|
||||
- If Yes, What is your Vulkan driver:
|
||||
|
||||
Add any other context about the problem here.
|
||||
@@ -13,14 +13,13 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_FORCE32BITALLOCATOR: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2], [self-hosted, ARMv8.4]]
|
||||
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
@@ -29,24 +28,9 @@ jobs:
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
run: git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
@@ -64,7 +48,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -132,18 +116,6 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
|
||||
|
||||
- name: gcc target tests 32
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
|
||||
|
||||
- name: GCC32 Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
|
||||
|
||||
- name: Struct verifier tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
@@ -155,39 +127,6 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
|
||||
|
||||
- name: APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target api_tests
|
||||
|
||||
- name: APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
|
||||
|
||||
- name: Thunkgen tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunkgen_tests
|
||||
|
||||
- name: Thunkgen Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
|
||||
@@ -10,4 +10,3 @@ out/
|
||||
.vscode/
|
||||
.vs/
|
||||
*.pyc
|
||||
.cache
|
||||
+1
-24
@@ -1,7 +1,7 @@
|
||||
[submodule "External/vixl"]
|
||||
shallow = true
|
||||
path = External/vixl
|
||||
url = https://github.com/FEX-Emu/vixl.git
|
||||
url = https://github.com/Sonicadvance1/vixl.git
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = External/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
@@ -30,26 +30,3 @@
|
||||
shallow = true
|
||||
path = External/fex-gcc-target-tests-bins
|
||||
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
|
||||
[submodule "External/jemalloc"]
|
||||
path = External/jemalloc
|
||||
url = https://github.com/FEX-Emu/jemalloc.git
|
||||
[submodule "External/fmt"]
|
||||
path = External/fmt
|
||||
url = https://github.com/fmtlib/fmt.git
|
||||
[submodule "External/drm-headers"]
|
||||
path = External/drm-headers
|
||||
url = https://github.com/FEX-Emu/drm-headers.git
|
||||
[submodule "External/xxhash"]
|
||||
path = External/xxhash
|
||||
url = https://github.com/FEX-Emu/xxHash.git
|
||||
[submodule "External/Catch2"]
|
||||
path = External/Catch2
|
||||
url = https://github.com/catchorg/Catch2.git
|
||||
[submodule "External/robin-map"]
|
||||
shallow = true
|
||||
path = External/robin-map
|
||||
url = https://github.com/Tessil/robin-map.git
|
||||
[submodule "External/Vulkan-Headers"]
|
||||
shallow = true
|
||||
path = External/Vulkan-Headers
|
||||
url = https://github.com/KhronosGroup/Vulkan-Headers.git
|
||||
+72
-273
@@ -1,41 +1,24 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
project(FEX)
|
||||
|
||||
INCLUDE (CheckIncludeFiles)
|
||||
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with lld" FALSE)
|
||||
option(ENABLE_MOLD "Enable linking with mold" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with LLD" FALSE)
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
|
||||
set (X86_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86.cmake" CACHE FILEPATH "Toolchain file for the x86 (cross-)compiler")
|
||||
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
|
||||
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
# These options are meant for package management
|
||||
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
|
||||
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
set (OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version in the format of <MMYY>{.<REV>}")
|
||||
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
|
||||
if (CMAKE_BUILD_TYPE MATCHES "DEBUG")
|
||||
set(ENABLE_ASSERTIONS TRUE)
|
||||
@@ -46,17 +29,6 @@ if (ENABLE_ASSERTIONS)
|
||||
add_definitions(-DASSERTIONS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_GDB_SYMBOLS)
|
||||
message(STATUS "GDBSymbols support enabled")
|
||||
add_definitions(-DGDB_SYMBOLS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
message(STATUS "Interpreter enabled")
|
||||
add_definitions(-DINTERPRETER_ENABLED=1)
|
||||
endif()
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
|
||||
@@ -72,6 +44,38 @@ else()
|
||||
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION FALSE)
|
||||
endif()
|
||||
|
||||
find_program(CCACHE_PROGRAM ccache)
|
||||
if(CCACHE_PROGRAM)
|
||||
message(STATUS "CCache enabled")
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
|
||||
endif()
|
||||
|
||||
if (ENABLE_XRAY)
|
||||
add_compile_options(-fxray-instrument)
|
||||
link_libraries(-fxray-instrument)
|
||||
endif()
|
||||
|
||||
if (ENABLE_LLD)
|
||||
link_libraries(-fuse-ld=lld)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
add_definitions(-DENABLE_ASAN=1)
|
||||
add_compile_options(-fno-omit-frame-pointer -fsanitize=address)
|
||||
link_libraries(-fno-omit-frame-pointer -fsanitize=address)
|
||||
endif()
|
||||
|
||||
if (ENABLE_TSAN)
|
||||
add_compile_options(-fno-omit-frame-pointer -fsanitize=thread)
|
||||
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
|
||||
endif()
|
||||
|
||||
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
|
||||
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
|
||||
if (NOT ENABLE_X86_HOST_DEBUG)
|
||||
@@ -84,7 +88,6 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(_M_X86_64 1)
|
||||
add_definitions(-D_M_X86_64=1)
|
||||
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
set (X86_TOOLCHAIN_FILE "")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
@@ -92,84 +95,6 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
add_definitions(-D_M_ARM_64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_CCACHE)
|
||||
find_program(CCACHE_PROGRAM ccache)
|
||||
if(CCACHE_PROGRAM)
|
||||
message(STATUS "CCache enabled")
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_XRAY)
|
||||
add_compile_options(-fxray-instrument)
|
||||
link_libraries(-fxray-instrument)
|
||||
endif()
|
||||
|
||||
if (ENABLE_COMPILE_TIME_TRACE)
|
||||
add_compile_options(-ftime-trace)
|
||||
link_libraries(-ftime-trace)
|
||||
endif()
|
||||
|
||||
set (PTHREAD_LIB pthread)
|
||||
|
||||
if (ENABLE_LLD AND ENABLE_MOLD)
|
||||
message (FATAL_ERROR "Cannot enable both lld and mold")
|
||||
elseif (ENABLE_LLD)
|
||||
set (LD_OVERRIDE "-fuse-ld=lld")
|
||||
add_link_options(${LD_OVERRIDE})
|
||||
elseif (ENABLE_MOLD)
|
||||
add_link_options("-fuse-ld=mold")
|
||||
endif()
|
||||
|
||||
if (ENABLE_LIBCXX)
|
||||
message(WARNING "This is an unsupported configuration and should only be used for testing")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -stdlib=libc++")
|
||||
set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -stdlib=libc++ -lc++abi")
|
||||
endif()
|
||||
|
||||
if (NOT ENABLE_OFFLINE_TELEMETRY)
|
||||
# Disable FEX offline telemetry entirely if asked
|
||||
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
|
||||
endif()
|
||||
|
||||
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
add_definitions(-DTERMUX_BUILD=1)
|
||||
set(TERMUX_BUILD 1)
|
||||
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
add_definitions(-DENABLE_ASAN=1)
|
||||
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
|
||||
link_libraries(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
|
||||
endif()
|
||||
|
||||
if (ENABLE_TSAN)
|
||||
add_compile_options(-fno-omit-frame-pointer -fsanitize=thread)
|
||||
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
add_definitions(-DENABLE_JEMALLOC=1)
|
||||
add_subdirectory(External/jemalloc/)
|
||||
include_directories(External/jemalloc/pregen/include/)
|
||||
else()
|
||||
message (STATUS
|
||||
" jemalloc disabled!\n"
|
||||
" This is not a recommended configuration!\n"
|
||||
" This will very explicitly break 32-bit application execution!\n"
|
||||
" Use at your own risk!")
|
||||
endif()
|
||||
|
||||
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
|
||||
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
|
||||
include_directories(External/robin-map/include/)
|
||||
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(External/vixl/src/)
|
||||
|
||||
@@ -178,30 +103,14 @@ if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
|
||||
endif()
|
||||
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
add_subdirectory(External/xxhash/)
|
||||
include_directories(External/xxhash/)
|
||||
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
option(CATCH_BUILD_STATIC_LIBRARY "" ON)
|
||||
set(CATCH_BUILD_STATIC_LIBRARY ON)
|
||||
add_subdirectory(External/Catch2/)
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/cpp-optparse/)
|
||||
include_directories(External/cpp-optparse/)
|
||||
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
include_directories(External/imgui/)
|
||||
|
||||
@@ -235,6 +144,11 @@ if(ENUM_ENUM_WARNING)
|
||||
add_compile_options(-Wno-deprecated-enum-enum-conversion)
|
||||
endif()
|
||||
|
||||
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -march=native")
|
||||
endif()
|
||||
|
||||
if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
add_compile_options(-Werror)
|
||||
if (NOT ENABLE_STRICT_WERROR)
|
||||
@@ -243,51 +157,19 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
add_compile_options("-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
endif()
|
||||
endif()
|
||||
if(_M_ARM_64)
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
|
||||
if (TUNE_CPU STREQUAL "native")
|
||||
if(_M_ARM_64)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
|
||||
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
|
||||
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
|
||||
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=native")
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
|
||||
|
||||
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
|
||||
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${AARCH64_CPU}")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
add_compile_options("-march=native")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${TUNE_CPU}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=${AARCH64_CPU}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -354,141 +236,58 @@ add_compile_options(-Wall)
|
||||
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/include/Config.h.in
|
||||
${CMAKE_BINARY_DIR}/generated/ConfigDefines.h)
|
||||
${CMAKE_BINARY_DIR}/generated/Config.h)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
include(CTest)
|
||||
enable_testing()
|
||||
message(STATUS "Unit tests are enabled")
|
||||
endif()
|
||||
|
||||
add_subdirectory(FEXHeaderUtils/)
|
||||
include_directories(FEXHeaderUtils/)
|
||||
|
||||
add_subdirectory(External/FEXCore)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
|
||||
add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
# Install the ThunksDB file
|
||||
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
|
||||
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
endforeach()
|
||||
|
||||
if (BUILD_TESTS)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
|
||||
if (BUILD_THUNKS)
|
||||
set (FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
|
||||
add_subdirectory(ThunkLibs/Generator)
|
||||
|
||||
# Thunk targets for both host libraries and IDE integration
|
||||
add_subdirectory(ThunkLibs/HostLibs)
|
||||
|
||||
# Thunk targets for IDE integration of guest code, only
|
||||
add_subdirectory(ThunkLibs/GuestLibs)
|
||||
|
||||
# Thunk targets for guest libraries
|
||||
include(ExternalProject)
|
||||
|
||||
ExternalProject_Add(host-libs
|
||||
PREFIX host-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/HostLibs"
|
||||
BINARY_DIR "Host"
|
||||
CMAKE_ARGS "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "MESSAGE(\"-- Installing: host-libs\")"
|
||||
CODE "
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkHostsInstall
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Host
|
||||
)"
|
||||
DEPENDS host-libs
|
||||
)
|
||||
|
||||
ExternalProject_Add(guest-libs
|
||||
PREFIX guest-libs
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
CMAKE_ARGS "-DX86_C_COMPILER:STRING=${X86_C_COMPILER}" "-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}" "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "MESSAGE(\"-- Installing: guest-libs\")"
|
||||
CODE "
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkGuestsInstall
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)"
|
||||
DEPENDS guest-libs
|
||||
)
|
||||
endif()
|
||||
|
||||
set(FEX_VERSION_MAJOR "0")
|
||||
set(FEX_VERSION_MINOR "0")
|
||||
set(FEX_VERSION_PATCH "0")
|
||||
|
||||
if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
find_package(Git)
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
RESULT_VARIABLE GIT_ERROR
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
|
||||
if (NOT ${GIT_ERROR} EQUAL 0)
|
||||
# Likely built in a way that doesn't have tags
|
||||
# Setup a version tag that is unknown
|
||||
set(GIT_DESCRIBE_STRING "FEX-0000")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
# Parse the version here
|
||||
# Change something like `FEX-2106.1-76-<hash>` in to a list
|
||||
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
|
||||
|
||||
# Extract the `2106.1` element
|
||||
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
|
||||
|
||||
# Change `2106.1` in to a list
|
||||
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
|
||||
|
||||
# Calculate list size
|
||||
list(LENGTH DESCRIBE_LIST LIST_SIZE)
|
||||
|
||||
# Pull out the major version
|
||||
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
|
||||
|
||||
# Minor version only exists if there is a .1 at the end
|
||||
# eg: 2106 versus 2106.1
|
||||
if (LIST_SIZE GREATER 1)
|
||||
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
|
||||
endif()
|
||||
|
||||
# Package creation
|
||||
set (CPACK_GENERATOR "DEB")
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
|
||||
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
|
||||
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/CPack/Description.txt")
|
||||
|
||||
# Debian defines
|
||||
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
|
||||
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/CPack/triggers")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
# binfmt_misc conflicts with qemu-user-static
|
||||
# We also only install binfmt_misc on aarch64 hosts
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
|
||||
endif()
|
||||
include (CPack)
|
||||
+1
-1
@@ -55,7 +55,7 @@ further defined and clarified by project maintainers.
|
||||
## Enforcement
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported by contacting the project team at team@fex-emu.com. All
|
||||
reported by contacting the project team at team@fex-emu.org. All
|
||||
complaints will be reviewed and investigated and will result in a response that
|
||||
is deemed necessary and appropriate to the circumstances. The project team is
|
||||
obligated to maintain confidentiality with regard to the reporter of an incident.
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
x86 and x86-64 Linux emulator
|
||||
|
||||
FEX is very much work in progress, so expect things to change.
|
||||
@@ -1,18 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Setup binfmt_misc
|
||||
update-binfmts --import FEX-x86
|
||||
update-binfmts --import FEX-x86_64
|
||||
}
|
||||
|
||||
# Install FEXInterpreter hardlink
|
||||
# Needs to be done before setting up binfmt_misc
|
||||
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
-17
@@ -1,17 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Uninstall
|
||||
update-binfmts --unimport FEX-x86
|
||||
update-binfmts --unimport FEX-x86_64
|
||||
}
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
|
||||
# Remove FEXInterpreter hardlink
|
||||
unlink /usr/bin/FEXInterpreter
|
||||
@@ -1 +0,0 @@
|
||||
activate-noawait ldconfig
|
||||
@@ -11,15 +11,15 @@ endforeach()
|
||||
# First generate then install it
|
||||
foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
|
||||
# Get the filename only component
|
||||
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WLE)
|
||||
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WE)
|
||||
|
||||
# Configure it
|
||||
configure_file(
|
||||
${GEN_CONFIG_SRC}
|
||||
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
|
||||
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json)
|
||||
|
||||
# Then install the configured json
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
endforeach()
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"Config": {
|
||||
"AdditionalArguments": "--no-sandbox"
|
||||
}
|
||||
}
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -1,170 +0,0 @@
|
||||
{
|
||||
"DB": {
|
||||
"GL": {
|
||||
"Library" : "libGL-guest.so",
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
"Library": "libGLESv2-guest.so",
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan": {
|
||||
"Library": "libvulkan-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so.1",
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
],
|
||||
"Comment": [
|
||||
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb_dri2-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb_dri3-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb_xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb_shm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb_sync-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb_randr-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb_present-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb_glx-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2.4.0"
|
||||
]
|
||||
},
|
||||
"asound": {
|
||||
"Library": "libasound-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
}
|
||||
}
|
||||
@@ -1,17 +0,0 @@
|
||||
function(GenBinFmt Name)
|
||||
# Get the filename only component
|
||||
get_filename_component(FMT_NAME ${Name} NAME_WE)
|
||||
|
||||
# Configure it
|
||||
configure_file(
|
||||
${Name}
|
||||
${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
|
||||
|
||||
# Then install the configured binfmt
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
|
||||
endfunction()
|
||||
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
@@ -1,8 +0,0 @@
|
||||
package fex
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
|
||||
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve yes
|
||||
@@ -1,8 +0,0 @@
|
||||
package fex
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
|
||||
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve yes
|
||||
+2
-2
@@ -3,7 +3,7 @@ FROM ubuntu:20.04 as builder
|
||||
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
|
||||
clang-10 llvm-10 nasm ninja-build \
|
||||
clang-10 llvm-10 nasm ninja-build libnuma-dev \
|
||||
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
|
||||
python3 linux-headers-generic
|
||||
|
||||
@@ -23,7 +23,7 @@ FROM ubuntu:20.04
|
||||
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt install -y \
|
||||
libcap-dev libglfw3-dev libepoxy-dev
|
||||
libnuma-dev libcap-dev libglfw3-dev libepoxy-dev
|
||||
|
||||
COPY --from=builder /opt/FEX/build/Bin/* /usr/bin/
|
||||
|
||||
|
||||
Vendored
-1
Submodule External/Catch2 deleted from c4e3767e26.
Vendored
+19
-23
@@ -16,6 +16,7 @@ endif()
|
||||
set(ENABLE_JIT_X86_64 ${_M_X86_64} CACHE BOOL "Enable the x86_64 JIT")
|
||||
set(ENABLE_JIT_ARM64 ${_M_ARM_64} CACHE BOOL "Enable the ARM64 JIT")
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_JITSYMBOLS "Enable visibility of JITSymbols in profiling tools" FALSE)
|
||||
|
||||
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
|
||||
cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
|
||||
@@ -37,32 +38,27 @@ endif()
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
|
||||
# Find our git hash
|
||||
find_package(Git)
|
||||
|
||||
set(GIT_SHORT_HASH "Unknown")
|
||||
set(GIT_DESCRIBE_STRING "FEX-Unknown")
|
||||
|
||||
if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
# Find our git hash
|
||||
find_package(Git)
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short=7 HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_SHORT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
endif()
|
||||
else()
|
||||
set(GIT_SHORT_HASH "${OVERRIDE_VERSION}")
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_SHORT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
endif()
|
||||
|
||||
configure_file(
|
||||
|
||||
Vendored
+12
-2
@@ -18,8 +18,14 @@ This project aims to provide a fast and functional x86-64 emulation library that
|
||||
* Portable library implementation in order to support easy integration in to applications
|
||||
### Target Host Architecture
|
||||
The target host architecture for this library is AArch64. Specifically the ARMv8.1 version or newer.
|
||||
The CPU IR is designed with AArch64 in mind but should allow for other architectures as well.
|
||||
x86-64 host support is available for ease of development, but is not a priority.
|
||||
The CPU IR is designed with AArch64 in mind but there is a desire to run the recompiled code on other architectures as well.
|
||||
Multiple architecture support is desired for easier bringup and debugging, performance isn't as much of a priority there (ex. x86-64(guest) translated to x86-64(host))
|
||||
### Not currently goals but will be in the future
|
||||
* 32bit x86 support
|
||||
* This will be a desire in the future, but to lower the amount of work required, decided to push this off for now.
|
||||
* Integration in to WINE
|
||||
* Later generation of x86-64 instruction sets
|
||||
* Including AVX, F16C, XOP, FMA, AVX2, etc
|
||||
### Not desired
|
||||
* Kernel space emulation
|
||||
* CPL0-2 emulation
|
||||
@@ -27,3 +33,7 @@ x86-64 host support is available for ease of development, but is not a priority.
|
||||
* IRQs
|
||||
* SVM
|
||||
* "Cycle Accurate" emulation
|
||||
### Dependencies
|
||||
* clang-tidy if you want to ensure the code stays tidy
|
||||
* cmake
|
||||
* A C++17 compliant compiler (There are assumptions made about using Clang and LTO)
|
||||
+5
-65
@@ -98,19 +98,14 @@ def print_man_option(short, long, desc, default):
|
||||
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
|
||||
output_man.write(".Pp\n\n")
|
||||
|
||||
def print_man_env_option(name, desc, default, no_json_key):
|
||||
output_man.write("\\fBFEX_{0}\\fR\n".format(name.upper()))
|
||||
def print_man_env_option(name, desc, default):
|
||||
output_man.write("\\fBFEX_{0}\\fR\n".format(name))
|
||||
|
||||
# Print description
|
||||
for line in desc:
|
||||
output_man.write(".Pp\n")
|
||||
output_man.write("{0}\n".format(line))
|
||||
|
||||
if (not no_json_key):
|
||||
output_man.write(".Pp\n")
|
||||
output_man.write("\\fBJSON key:\\fR '{0}'\n".format(name))
|
||||
output_man.write(".Pp\n\n")
|
||||
|
||||
output_man.write(".Pp\n")
|
||||
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
|
||||
output_man.write(".Pp\n\n")
|
||||
@@ -159,48 +154,12 @@ def print_man_environment(options):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_env_option(
|
||||
op_key,
|
||||
op_key.upper(),
|
||||
op_vals["Desc"],
|
||||
default,
|
||||
False
|
||||
default
|
||||
)
|
||||
|
||||
print_man_environment_tail()
|
||||
output_man.write(".El\n")
|
||||
|
||||
def print_man_environment_tail():
|
||||
|
||||
# Additional environment variables that live outside of the normal loop
|
||||
print_man_env_option(
|
||||
"FEX_APP_CONFIG_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for configuration files",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_APP_CONFIG",
|
||||
[
|
||||
"Allows the user to override where FEX looks for only the application config file",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/Config.json",
|
||||
"This will override this file location",
|
||||
"One must be careful with this option as it will override any applications that load with execve as well"
|
||||
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_APP_DATA_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for data files",
|
||||
"By default FEX will look in {$HOME, $XDG_DATA_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
"This is the folder where FEX stores generated files like IR cache"
|
||||
],
|
||||
"''", True)
|
||||
|
||||
def print_man_header():
|
||||
header ='''.Dd {0}
|
||||
.Dt FEX
|
||||
@@ -374,7 +333,7 @@ def print_parse_argloader_options(options):
|
||||
conversion_func = "std::to_string"
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
NeedsString = True
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
conversion_func = "FEX::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = ""
|
||||
@@ -396,21 +355,6 @@ def print_parse_argloader_options(options):
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
|
||||
def print_parse_envloader_options(options):
|
||||
output_argloader.write("#ifdef ENVLOADER\n")
|
||||
output_argloader.write("#undef ENVLOADER\n")
|
||||
output_argloader.write("if (false) {}\n")
|
||||
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
output_argloader.write("Value = {0}(Value);\n".format(conversion_func))
|
||||
output_argloader.write("}\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def check_for_duplicate_options(options):
|
||||
short_map = []
|
||||
long_map = []
|
||||
@@ -485,8 +429,4 @@ output_man.close()
|
||||
output_argloader = open(output_argumentloader_filename, "w")
|
||||
print_argloader_options(options);
|
||||
print_parse_argloader_options(options);
|
||||
|
||||
# Generate environment loader code
|
||||
print_parse_envloader_options(options);
|
||||
|
||||
output_argloader.close()
|
||||
+43
-6
@@ -7,12 +7,17 @@ OpClasses = collections.OrderedDict()
|
||||
def get_ir_classes(ops, defines):
|
||||
global OpClasses
|
||||
|
||||
for op_class, opslist in ops.items():
|
||||
if not (op_class in OpClasses):
|
||||
OpClasses[op_class] = []
|
||||
for op_key, op_vals in ops.items():
|
||||
if not ("Last" in op_vals):
|
||||
OpClass = "#Unknown"
|
||||
|
||||
for op, op_val in opslist.items():
|
||||
OpClasses[op_class].append([op, op_val])
|
||||
if ("OpClass" in op_vals):
|
||||
OpClass = op_vals["OpClass"]
|
||||
|
||||
if not (OpClass in OpClasses):
|
||||
OpClasses[OpClass] = []
|
||||
|
||||
OpClasses[OpClass].append([op_key, op_vals])
|
||||
|
||||
# Sort the dictionary after we are done parsing it
|
||||
OpClasses = collections.OrderedDict(sorted(OpClasses.items()))
|
||||
@@ -33,9 +38,41 @@ def print_ir_ops():
|
||||
op_key = op[0]
|
||||
op_vals = op[1]
|
||||
output_file.write("## %s\n" % (op_key))
|
||||
HasDest = ("HasDest" in op_vals and op_vals["HasDest"] == True)
|
||||
HasSSAArgs = ("SSAArgs" in op_vals and len(op_vals["SSAArgs"]) > 0)
|
||||
HasSSAArgNames = "SSANames" in op_vals
|
||||
HasArgs = "Args" in op_vals
|
||||
SSAArgsCount = 0
|
||||
ArgCount = 0
|
||||
if (HasSSAArgs):
|
||||
SSAArgsCount = int(op_vals["SSAArgs"])
|
||||
if (HasArgs):
|
||||
ArgCount = len(op_vals["Args"])
|
||||
|
||||
TotalArgsCount = SSAArgsCount + (ArgCount / 2)
|
||||
|
||||
output_file.write(">")
|
||||
output_file.write(op_key)
|
||||
if (HasDest):
|
||||
output_file.write("%dest = ")
|
||||
|
||||
output_file.write("%s " % op_key)
|
||||
|
||||
ArgComma = (", ", "")
|
||||
if (HasSSAArgs):
|
||||
for i in range(0, SSAArgsCount):
|
||||
FinalArg = (i + 1) == TotalArgsCount
|
||||
if (HasSSAArgNames):
|
||||
output_file.write("%%%s%s" % (op_vals["SSANames"][i], ArgComma[FinalArg]))
|
||||
else:
|
||||
output_file.write("%%ssa%d%s" % (i, ArgComma[FinalArg]))
|
||||
|
||||
if (HasArgs):
|
||||
Args = op_vals["Args"]
|
||||
for i in range(0, ArgCount, 2):
|
||||
FinalArg = ((i / 2) + SSAArgsCount + 1) == TotalArgsCount
|
||||
data_type = Args[i]
|
||||
data_name = Args[i + 1]
|
||||
output_file.write("\<%s %s\>%s" % (data_type, data_name, ArgComma[FinalArg]))
|
||||
|
||||
output_file.write("\n\n")
|
||||
|
||||
|
||||
+395
-492
File diff suppressed because it is too large.
Load diff
+37
-110
@@ -1,15 +1,9 @@
|
||||
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (FEXCORE_BASE_SRCS
|
||||
Common/Paths.cpp
|
||||
Interface/Config/Config.cpp
|
||||
Utils/FileLoading.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
)
|
||||
|
||||
set (SRCS
|
||||
Common/Paths.cpp
|
||||
Common/JitSymbols.cpp
|
||||
Common/NetStream.cpp
|
||||
Common/SoftFloat-3e/extF80_add.c
|
||||
Common/SoftFloat-3e/extF80_div.c
|
||||
Common/SoftFloat-3e/extF80_sub.c
|
||||
@@ -77,25 +71,17 @@ set (SRCS
|
||||
Common/SoftFloat-3e/f32_to_extF80.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF32Sig.c
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
Interface/Config/Config.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
Interface/Core/CompileService.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/GdbServer.cpp
|
||||
Interface/Core/HostFeatures.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
Interface/Core/ObjectCache/ObjectCacheService.cpp
|
||||
Interface/Core/OpcodeDispatcher/Crypto.cpp
|
||||
Interface/Core/OpcodeDispatcher/Flags.cpp
|
||||
Interface/Core/OpcodeDispatcher/Vector.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/SignalDelegator.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
Interface/Core/X86DebugInfo.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
@@ -104,7 +90,8 @@ set (SRCS
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/X86Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
|
||||
Interface/Core/Interpreter/InterpreterFallbacks.cpp
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
Interface/Core/Interpreter/InterpreterOps.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
@@ -118,8 +105,6 @@ set (SRCS
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IRParser.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
@@ -129,8 +114,6 @@ set (SRCS
|
||||
Interface/IR/Passes/DeadContextStoreElimination.cpp
|
||||
Interface/IR/Passes/IRCompaction.cpp
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RAValidation.cpp
|
||||
Interface/IR/Passes/LongDivideRemovalPass.cpp
|
||||
Interface/IR/Passes/ValueDominanceValidation.cpp
|
||||
Interface/IR/Passes/PhiValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
@@ -138,36 +121,18 @@ set (SRCS
|
||||
Interface/IR/Passes/StaticRegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/Allocator/64BitAllocator.cpp
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/ELFLoader.cpp
|
||||
Utils/ELFSymbolDatabase.cpp
|
||||
Utils/LogManager.cpp
|
||||
Utils/Threads.cpp
|
||||
)
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
Interface/Core/Interpreter/InterpreterOps.cpp
|
||||
Interface/Core/Interpreter/ALUOps.cpp
|
||||
Interface/Core/Interpreter/AtomicOps.cpp
|
||||
Interface/Core/Interpreter/BranchOps.cpp
|
||||
Interface/Core/Interpreter/ConversionOps.cpp
|
||||
Interface/Core/Interpreter/EncryptionOps.cpp
|
||||
Interface/Core/Interpreter/F80Ops.cpp
|
||||
Interface/Core/Interpreter/FlagOps.cpp
|
||||
Interface/Core/Interpreter/MemoryOps.cpp
|
||||
Interface/Core/Interpreter/MiscOps.cpp
|
||||
Interface/Core/Interpreter/MoveOps.cpp
|
||||
Interface/Core/Interpreter/VectorOps.cpp)
|
||||
endif()
|
||||
|
||||
if(_M_ARM_64)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/ArchHelpers/Arm64.cpp)
|
||||
endif()
|
||||
|
||||
set(DEFINES -DTHREAD_LOCAL=_Thread_local)
|
||||
set(DEFINES )
|
||||
|
||||
if (_M_X86_64)
|
||||
list(APPEND DEFINES -D_M_X86_64=1)
|
||||
@@ -189,9 +154,7 @@ if (ENABLE_JIT_X86_64)
|
||||
Interface/Core/JIT/x86_64/MemoryOps.cpp
|
||||
Interface/Core/JIT/x86_64/MiscOps.cpp
|
||||
Interface/Core/JIT/x86_64/MoveOps.cpp
|
||||
Interface/Core/JIT/x86_64/VectorOps.cpp
|
||||
Interface/Core/JIT/x86_64/x64Relocations.cpp
|
||||
)
|
||||
Interface/Core/JIT/x86_64/VectorOps.cpp)
|
||||
list(APPEND DEFINES -DJIT_X86_64)
|
||||
endif()
|
||||
|
||||
@@ -208,31 +171,25 @@ if (ENABLE_JIT_ARM64)
|
||||
Interface/Core/JIT/Arm64/MemoryOps.cpp
|
||||
Interface/Core/JIT/Arm64/MiscOps.cpp
|
||||
Interface/Core/JIT/Arm64/MoveOps.cpp
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp
|
||||
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
|
||||
)
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp)
|
||||
endif()
|
||||
|
||||
set (LIBS vixl dl xxhash tiny-json)
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
if (ENABLE_JITSYMBOLS)
|
||||
list(APPEND DEFINES -DENABLE_JITSYMBOLS=1)
|
||||
endif()
|
||||
|
||||
# Generate config
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
${CMAKE_BINARY_DIR}/generated/Config/Config.json)
|
||||
|
||||
# Generate IR include file
|
||||
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
|
||||
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
|
||||
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
|
||||
|
||||
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
|
||||
add_custom_target(CREATE_IR_FOLDER ALL
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_IR_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_NAME}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS CREATE_IR_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
|
||||
)
|
||||
@@ -246,6 +203,7 @@ set(OUTPUT_IR_DOC "${CMAKE_BINARY_DIR}/IR.md")
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_IR_DOC}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS CREATE_IR_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py" "${INPUT_NAME}" "${OUTPUT_IR_DOC}"
|
||||
)
|
||||
@@ -262,28 +220,23 @@ add_custom_target(IR_INC
|
||||
set(OUTPUT_CONFIG_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/Config")
|
||||
set(OUTPUT_CONFIG_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigValues.inl")
|
||||
set(OUTPUT_CONFIG_OPTION_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigOptions.inl")
|
||||
set(INPUT_CONFIG_NAME "${CMAKE_BINARY_DIR}/generated/Config/Config.json")
|
||||
set(INPUT_CONFIG_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json")
|
||||
set(OUTPUT_MAN_NAME "${CMAKE_BINARY_DIR}/generated/FEX.1")
|
||||
set(OUTPUT_MAN_NAME_COMPRESS "${CMAKE_BINARY_DIR}/generated/FEX.1.gz")
|
||||
|
||||
file(MAKE_DIRECTORY "${OUTPUT_CONFIG_FOLDER}")
|
||||
add_custom_target(CREATE_CONFIG_FOLDER ALL
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_CONFIG_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_CONFIG_NAME}"
|
||||
OUTPUT "${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
OUTPUT "${OUTPUT_MAN_NAME}"
|
||||
DEPENDS "${INPUT_CONFIG_NAME}"
|
||||
DEPENDS CREATE_CONFIG_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py" "${INPUT_CONFIG_NAME}" "${OUTPUT_CONFIG_NAME}" "${OUTPUT_MAN_NAME}"
|
||||
"${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
)
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_MAN_NAME_COMPRESS}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME}"
|
||||
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}"
|
||||
)
|
||||
|
||||
set_source_files_properties(${OUTPUT_CONFIG_NAME} PROPERTIES
|
||||
GENERATED TRUE)
|
||||
set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
|
||||
@@ -291,28 +244,29 @@ set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
|
||||
|
||||
set_source_files_properties(${OUTPUT_MAN_NAME} PROPERTIES
|
||||
GENERATED TRUE)
|
||||
set_source_files_properties(${OUTPUT_MAN_NAME_COMPRESS} PROPERTIES
|
||||
GENERATED TRUE)
|
||||
|
||||
# Create the target
|
||||
add_custom_target(CONFIG_INC
|
||||
DEPENDS "${OUTPUT_CONFIG_NAME}"
|
||||
DEPENDS "${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
|
||||
DEPENDS "${OUTPUT_MAN_NAME}")
|
||||
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
|
||||
# Install the man page
|
||||
install(FILES ${OUTPUT_MAN_NAME} DESTINATION ${MAN_DIR}/man1)
|
||||
|
||||
# Add in diagnostic colours if the option is available.
|
||||
# Ninja code generator will kill colours if this isn't here
|
||||
check_cxx_compiler_flag(-fdiagnostics-color=always GCC_COLOR)
|
||||
check_cxx_compiler_flag(-fcolor-diagnostics CLANG_COLOR)
|
||||
|
||||
function(AddDefaultOptionsToTarget Name)
|
||||
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES VISIBILITY_INLINES_HIDDEN TRUE)
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
add_dependencies(${Name} CONFIG_INC)
|
||||
|
||||
target_link_libraries(${Name} pthread rt vixl ${LINUX_LIBS} dl)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
|
||||
|
||||
target_include_directories(${Name} PRIVATE IncludePrivate/)
|
||||
@@ -322,18 +276,13 @@ function(AddDefaultOptionsToTarget Name)
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
|
||||
|
||||
target_compile_definitions(${Name} PRIVATE ${DEFINES})
|
||||
add_dependencies(${Name} CONFIG_INC)
|
||||
|
||||
target_compile_options(${Name}
|
||||
PRIVATE
|
||||
-Wall
|
||||
-Werror=cast-qual
|
||||
-Werror=ignored-qualifiers
|
||||
-Werror=implicit-fallthrough
|
||||
|
||||
-Wno-trigraphs
|
||||
-ffunction-sections
|
||||
-fwrapv
|
||||
)
|
||||
|
||||
if (GCC_COLOR)
|
||||
@@ -346,38 +295,16 @@ function(AddDefaultOptionsToTarget Name)
|
||||
PRIVATE
|
||||
"-fcolor-diagnostics")
|
||||
endif()
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
target_link_options(${Name}
|
||||
PRIVATE
|
||||
"LINKER:--gc-sections"
|
||||
"LINKER:--strip-all"
|
||||
"LINKER:--as-needed"
|
||||
)
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
# Build FEXCore_Config static library
|
||||
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base fmt::fmt tiny-json)
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
target_link_libraries(${Name} pthread rt vixl ${LINUX_LIBS} dl)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
|
||||
target_include_directories(${Name} PUBLIC "${PROJECT_SOURCE_DIR}/include/")
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
|
||||
endfunction()
|
||||
|
||||
AddObject(${PROJECT_NAME}_object OBJECT)
|
||||
|
||||
+13
-18
@@ -1,16 +1,12 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include "Common/MathUtils.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <type_traits>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
template<typename T>
|
||||
struct BitSet final {
|
||||
using ElementType = T;
|
||||
@@ -20,16 +16,16 @@ struct BitSet final {
|
||||
ElementType *Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
|
||||
LogMan::Throw::A((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(malloc(AllocateSize));
|
||||
}
|
||||
void Realloc(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
|
||||
LogMan::Throw::A((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(realloc(Memory, AllocateSize));
|
||||
}
|
||||
void Free() {
|
||||
FEXCore::Allocator::free(Memory);
|
||||
free(Memory);
|
||||
Memory = nullptr;
|
||||
}
|
||||
bool Get(T Element) {
|
||||
@@ -64,8 +60,8 @@ struct BitSetView final {
|
||||
ElementType *Memory;
|
||||
|
||||
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0,
|
||||
"Bitset view offset needs to be aligned to size of backing element");
|
||||
LogMan::Throw::A((ElementOffset % MinimumSize) == 0,
|
||||
"Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
@@ -90,12 +86,11 @@ struct BitSetView final {
|
||||
bool operator[](T Element) {
|
||||
return Get(Element);
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
static_assert(sizeof(BitSet<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
|
||||
static_assert(std::is_trivially_copyable_v<BitSet<uint32_t>>, "Needs to trivially copyable");
|
||||
static_assert(std::is_trivially_copyable<BitSet<uint32_t>>::value, "Needs to trivially copyable");
|
||||
|
||||
static_assert(sizeof(BitSetView<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
|
||||
static_assert(std::is_trivially_copyable_v<BitSetView<uint32_t>>, "Needs to trivially copyable");
|
||||
|
||||
} // namespace FEXCore
|
||||
static_assert(std::is_trivially_copyable<BitSetView<uint32_t>>::value, "Needs to trivially copyable");
|
||||
+21
-40
@@ -1,64 +1,45 @@
|
||||
#include "Common/JitSymbols.h"
|
||||
|
||||
#include <string>
|
||||
#include <sstream>
|
||||
#include <unistd.h>
|
||||
|
||||
#include <fmt/format.h>
|
||||
|
||||
namespace FEXCore {
|
||||
JITSymbols::JITSymbols() : fp{nullptr, std::fclose} {
|
||||
}
|
||||
JITSymbols::JITSymbols() {
|
||||
std::stringstream PerfMap;
|
||||
PerfMap << "/tmp/perf-" << getpid() << ".map";
|
||||
|
||||
JITSymbols::~JITSymbols() = default;
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
fp.reset(fopen(PerfMap.c_str(), "wb"));
|
||||
fp = fopen(PerfMap.str().c_str(), "wb");
|
||||
if (fp) {
|
||||
// Disable buffering on this file
|
||||
setvbuf(fp.get(), nullptr, _IONBF, 0);
|
||||
setvbuf(fp, nullptr, _IONBF, 0);
|
||||
}
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
JITSymbols::~JITSymbols() {
|
||||
if (fp) {
|
||||
fclose(fp);
|
||||
}
|
||||
}
|
||||
|
||||
void JITSymbols::Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
std::stringstream String;
|
||||
String << std::hex << HostAddr << " " << CodeSize << " " << "JIT_0x" << GuestAddr << "_" << HostAddr << std::endl;
|
||||
fwrite(String.str().c_str(), 1, String.str().size(), fp);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
void JITSymbols::Register(void *HostAddr, uint32_t CodeSize, std::string const &Name) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
std::stringstream String;
|
||||
String << std::hex << HostAddr << " " << CodeSize << " " << Name << "_" << HostAddr << std::endl;
|
||||
fwrite(String.str().c_str(), 1, String.str().size(), fp);
|
||||
}
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
+4
-13
@@ -1,26 +1,17 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <memory>
|
||||
#include <string_view>
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore {
|
||||
class JITSymbols final {
|
||||
public:
|
||||
JITSymbols();
|
||||
~JITSymbols();
|
||||
|
||||
void InitFile();
|
||||
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
|
||||
void Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(void *HostAddr, uint32_t CodeSize, std::string const &Name);
|
||||
|
||||
private:
|
||||
using FILEPtr = std::unique_ptr<FILE, decltype(&std::fclose)>;
|
||||
|
||||
FILEPtr fp;
|
||||
FILE* fp{};
|
||||
};
|
||||
}
|
||||
+13
@@ -0,0 +1,13 @@
|
||||
#pragma once
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
static inline uint64_t AlignUp(uint64_t value, uint64_t size) {
|
||||
return value + (size - value % size) % size;
|
||||
};
|
||||
|
||||
static inline uint64_t AlignDown(uint64_t value, uint64_t size) {
|
||||
return value - value % size;
|
||||
};
|
||||
|
||||
|
||||
+16
-45
@@ -1,47 +1,19 @@
|
||||
#include <FEXCore/Utils/NetStream.h>
|
||||
#include "NetStream.h"
|
||||
|
||||
#include <array>
|
||||
#include <cstring>
|
||||
#include <iterator>
|
||||
|
||||
#include <sys/types.h>
|
||||
#include <sys/socket.h>
|
||||
|
||||
#include <stdio.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
namespace {
|
||||
class NetBuf final : public std::streambuf {
|
||||
public:
|
||||
explicit NetBuf(int socketfd) : socket{socketfd} {
|
||||
reset_output_buffer();
|
||||
}
|
||||
~NetBuf() override {
|
||||
close(socket);
|
||||
}
|
||||
|
||||
private:
|
||||
std::streamsize xsputn(const char* buffer, std::streamsize size) override;
|
||||
|
||||
std::streambuf::int_type underflow() override;
|
||||
std::streambuf::int_type overflow(std::streambuf::int_type ch) override;
|
||||
int sync() override;
|
||||
|
||||
void reset_output_buffer() {
|
||||
// we always leave room for one extra char
|
||||
setp(std::begin(output_buffer), std::end(output_buffer) -1);
|
||||
}
|
||||
|
||||
int flushBuffer(const char *buffer, size_t size);
|
||||
|
||||
int socket;
|
||||
std::array<char, 1400> output_buffer;
|
||||
std::array<char, 1500> input_buffer; // enough for a typical packet
|
||||
};
|
||||
|
||||
int NetBuf::flushBuffer(const char *buffer, size_t size) {
|
||||
int NetStream::NetBuf::flushBuffer(const char *buffer, size_t size) {
|
||||
size_t total = 0;
|
||||
|
||||
// Send data
|
||||
while (total < size) {
|
||||
size_t sent = send(socket, (const void*)(buffer + total), size - total, MSG_NOSIGNAL);
|
||||
size_t sent = send(socket, (const void*)(buffer + total), size - total, 0);
|
||||
if (sent == -1) {
|
||||
// lets just assume all errors are end of file.
|
||||
return -1;
|
||||
@@ -52,12 +24,12 @@ int NetBuf::flushBuffer(const char *buffer, size_t size) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::streamsize NetBuf::xsputn(const char* buffer, std::streamsize size) {
|
||||
std::streamsize NetStream::NetBuf::xsputn(const char* buffer, std::streamsize size) {
|
||||
size_t buf_remaining = epptr() - pptr();
|
||||
|
||||
// Check if the string fits neatly in our buffer
|
||||
if (size <= buf_remaining) {
|
||||
::memcpy(pptr(), buffer, size);
|
||||
std::memcpy(pptr(), buffer, size);
|
||||
pbump(size);
|
||||
return size;
|
||||
}
|
||||
@@ -76,23 +48,23 @@ std::streamsize NetBuf::xsputn(const char* buffer, std::streamsize size) {
|
||||
}
|
||||
}
|
||||
|
||||
std::streambuf::int_type NetBuf::overflow(std::streambuf::int_type ch) {
|
||||
std::streambuf::int_type NetStream::NetBuf::overflow(std::streambuf::int_type ch) {
|
||||
// we always leave room for one extra char
|
||||
*pptr() = (char) ch;
|
||||
pbump(1);
|
||||
return sync();
|
||||
}
|
||||
|
||||
int NetBuf::sync() {
|
||||
int NetStream::NetBuf::sync() {
|
||||
// Flush and reset output buffer to zero
|
||||
if (flushBuffer(pbase(), pptr() - pbase()) < 0) {
|
||||
if(flushBuffer(pbase(), pptr() - pbase()) < 0) {
|
||||
return -1;
|
||||
}
|
||||
reset_output_buffer();
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::streambuf::int_type NetBuf::underflow() {
|
||||
std::streambuf::int_type NetStream::NetBuf::underflow() {
|
||||
ssize_t size = recv(socket, (void *)std::begin(input_buffer), sizeof(input_buffer), 0);
|
||||
|
||||
if (size <= 0) {
|
||||
@@ -104,12 +76,11 @@ std::streambuf::int_type NetBuf::underflow() {
|
||||
|
||||
return traits_type::to_int_type(*gptr());
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
NetStream::NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
|
||||
|
||||
NetStream::~NetStream() {
|
||||
delete rdbuf();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Utils
|
||||
NetStream::NetBuf::~NetBuf() {
|
||||
close(socket);
|
||||
}
|
||||
+41
@@ -0,0 +1,41 @@
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <iostream>
|
||||
#include <string.h>
|
||||
|
||||
class NetStream : public std::iostream {
|
||||
public:
|
||||
NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
|
||||
virtual ~NetStream();
|
||||
|
||||
private:
|
||||
class NetBuf : public std::streambuf {
|
||||
|
||||
public:
|
||||
NetBuf(int socketfd) {
|
||||
socket = socketfd;
|
||||
reset_output_buffer();
|
||||
}
|
||||
virtual ~NetBuf();
|
||||
|
||||
protected:
|
||||
virtual std::streamsize xsputn(const char* buffer, std::streamsize size);
|
||||
|
||||
virtual std::streambuf::int_type underflow();
|
||||
virtual std::streambuf::int_type overflow(std::streambuf::int_type ch);
|
||||
virtual int sync();
|
||||
|
||||
private:
|
||||
void reset_output_buffer() {
|
||||
// we always leave room for one extra char
|
||||
setp(std::begin(output_buffer), std::end(output_buffer) -1);
|
||||
}
|
||||
|
||||
int flushBuffer(const char *buffer, size_t size);
|
||||
|
||||
int socket;
|
||||
std::array<char, 1400> output_buffer;
|
||||
std::array<char, 1500> input_buffer; // enough for a typical packet
|
||||
};
|
||||
};
|
||||
+12
-53
@@ -3,48 +3,13 @@
|
||||
|
||||
#include <cstdlib>
|
||||
#include <filesystem>
|
||||
#include <memory>
|
||||
#include <pwd.h>
|
||||
#include <system_error>
|
||||
#include <unistd.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
namespace FEXCore::Paths {
|
||||
std::unique_ptr<std::string> CachePath;
|
||||
std::unique_ptr<std::string> EntryCache;
|
||||
|
||||
char const* FindUserHomeThroughUID() {
|
||||
auto passwd = getpwuid(geteuid());
|
||||
if (passwd) {
|
||||
return passwd->pw_dir;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const char *GetHomeDirectory() {
|
||||
char const *HomeDir = getenv("HOME");
|
||||
|
||||
// Try to get home directory from uid
|
||||
if (!HomeDir) {
|
||||
HomeDir = FindUserHomeThroughUID();
|
||||
}
|
||||
|
||||
// try the PWD
|
||||
if (!HomeDir) {
|
||||
HomeDir = getenv("PWD");
|
||||
}
|
||||
|
||||
// Still doesn't exit? You get local
|
||||
if (!HomeDir) {
|
||||
HomeDir = ".";
|
||||
}
|
||||
|
||||
return HomeDir;
|
||||
}
|
||||
std::string CachePath;
|
||||
std::string EntryCache;
|
||||
|
||||
void InitializePaths() {
|
||||
CachePath = std::make_unique<std::string>();
|
||||
EntryCache = std::make_unique<std::string>();
|
||||
|
||||
char const *HomeDir = getenv("HOME");
|
||||
|
||||
if (!HomeDir) {
|
||||
@@ -57,35 +22,29 @@ namespace FEXCore::Paths {
|
||||
|
||||
char *XDGDataDir = getenv("XDG_DATA_DIR");
|
||||
if (XDGDataDir) {
|
||||
*CachePath = XDGDataDir;
|
||||
CachePath = XDGDataDir;
|
||||
}
|
||||
else {
|
||||
if (HomeDir) {
|
||||
*CachePath = HomeDir;
|
||||
CachePath = HomeDir;
|
||||
}
|
||||
}
|
||||
|
||||
*CachePath += "/.fex-emu/";
|
||||
*EntryCache = *CachePath + "/EntryCache/";
|
||||
CachePath += "/.fex-emu/";
|
||||
EntryCache = CachePath + "/EntryCache/";
|
||||
|
||||
std::error_code ec{};
|
||||
// Ensure the folder structure is created for our Data
|
||||
if (!std::filesystem::exists(*EntryCache, ec) &&
|
||||
!std::filesystem::create_directories(*EntryCache, ec)) {
|
||||
LogMan::Msg::DFmt("Couldn't create EntryCache directory: '{}'", *EntryCache);
|
||||
if (!std::filesystem::exists(EntryCache) &&
|
||||
!std::filesystem::create_directories(EntryCache)) {
|
||||
LogMan::Msg::D("Couldn't create EntryCache directory: '%s'", EntryCache.c_str());
|
||||
}
|
||||
}
|
||||
|
||||
void ShutdownPaths() {
|
||||
CachePath.reset();
|
||||
EntryCache.reset();
|
||||
}
|
||||
|
||||
std::string GetCachePath() {
|
||||
return *CachePath;
|
||||
return CachePath;
|
||||
}
|
||||
|
||||
std::string GetEntryCachePath() {
|
||||
return *EntryCache;
|
||||
return EntryCache;
|
||||
}
|
||||
}
|
||||
-4
@@ -3,10 +3,6 @@
|
||||
|
||||
namespace FEXCore::Paths {
|
||||
void InitializePaths();
|
||||
void ShutdownPaths();
|
||||
|
||||
const char *GetHomeDirectory();
|
||||
|
||||
std::string GetCachePath();
|
||||
std::string GetEntryCachePath();
|
||||
}
|
||||
+27
-315
@@ -1,6 +1,4 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cmath>
|
||||
@@ -16,19 +14,9 @@ extern "C" {
|
||||
|
||||
struct X80SoftFloat {
|
||||
#ifdef _M_X86_64
|
||||
// Define this to push some operations to x87
|
||||
// Only useful to see if precision loss is killing something
|
||||
// #define DEBUG_X86_FLOAT
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
#define BIGFLOAT long double
|
||||
#define BIGFLOATSIZE 10
|
||||
#else
|
||||
#define BIGFLOAT __float128
|
||||
#define BIGFLOATSIZE 16
|
||||
#endif
|
||||
#elif defined(_M_ARM_64)
|
||||
#define BIGFLOAT long double
|
||||
#define BIGFLOATSIZE 16
|
||||
#else
|
||||
#error No 128bit float for this target!
|
||||
#endif
|
||||
@@ -57,183 +45,51 @@ struct X80SoftFloat {
|
||||
|
||||
// Ops
|
||||
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
faddp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_add(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fsubp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sub(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fmulp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_mul(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fdivp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_div(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fprem;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
X80SoftFloat Rem = extF80_rem(lhs, rhs);
|
||||
if (SignBit(Rem)) {
|
||||
Rem = extF80_add(Rem, rhs);
|
||||
}
|
||||
else {
|
||||
Rem.Sign = SignBit(lhs);
|
||||
}
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
#endif
|
||||
return Rem;
|
||||
}
|
||||
|
||||
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fprem1;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
|
||||
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
|
||||
}
|
||||
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
|
||||
return extF80_roundToInt(lhs, RoundMode, false);
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Tmp = lhs;
|
||||
Tmp.Exponent = 0x3FFF;
|
||||
Tmp.Sign = lhs.Sign;
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
int32_t TrueExp = lhs.Exponent - ExponentBias;
|
||||
return i32_to_extF80(TrueExp);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
|
||||
@@ -243,211 +99,77 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fscale; # st0 = st0 * 2^(rdint(st1))
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Int = FRNDINT(rhs, softfloat_round_minMag);
|
||||
WARN_ONCE("x87: Application used FSCALE which may have accuracy problems");
|
||||
X80SoftFloat Int = FRNDINT(rhs);
|
||||
BIGFLOAT Src2_d = Int;
|
||||
Src2_d = exp2l(Src2_d);
|
||||
X80SoftFloat Src2_X80 = Src2_d;
|
||||
X80SoftFloat Result = extF80_mul(lhs, Src2_X80);
|
||||
return Result;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
f2xm1; # st0 = 2^st(0) - 1
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
WARN_ONCE("x87: Application used F2XM1 which may have accuracy problems");
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Result = exp2l(Src1_d);
|
||||
Result -= 1.0;
|
||||
return Result;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st(1)
|
||||
fldt %[lhs]; # st(0)
|
||||
fyl2x; # st(1) * log2l(st(0))
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
WARN_ONCE("x87: Application used FYL2X which may have accuracy problems");
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = Src2_d * log2l(Src1_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs];
|
||||
fldt %[rhs];
|
||||
fpatan;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
WARN_ONCE("x87: Application used FATAN which may have accuracy problems");
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = atan2l(Src1_d, Src2_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fptan;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
WARN_ONCE("x87: Application used FTAN which may have accuracy problems");
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = tanl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsin;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
WARN_ONCE("x87: Application used FSIN which may have accuracy problems");
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = sinl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fcos;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
WARN_ONCE("x87: Application used FCOS which may have accuracy problems");
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = cosl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsqrt;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sqrt(lhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
operator float() const {
|
||||
const float32_t Result = extF80_to_f32(*this);
|
||||
return FEXCore::BitCast<float>(Result);
|
||||
float32_t Result = extF80_to_f32(*this);
|
||||
return *(float*)&Result;
|
||||
}
|
||||
|
||||
operator double() const {
|
||||
const float64_t Result = extF80_to_f64(*this);
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
float64_t Result = extF80_to_f64(*this);
|
||||
return *(double*)&Result;
|
||||
}
|
||||
|
||||
operator BIGFLOAT() const {
|
||||
#if BIGFLOATSIZE == 16
|
||||
const float128_t Result = extF80_to_f128(*this);
|
||||
return FEXCore::BitCast<BIGFLOAT>(Result);
|
||||
#else
|
||||
BIGFLOAT result{};
|
||||
memcpy(&result, this, sizeof(result));
|
||||
return result;
|
||||
#endif
|
||||
float128_t Result = extF80_to_f128(*this);
|
||||
return *(BIGFLOAT*)&Result;
|
||||
}
|
||||
|
||||
operator int16_t() const {
|
||||
@@ -474,11 +196,11 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
void operator=(const float rhs) {
|
||||
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
|
||||
*this = f32_to_extF80(*(float32_t*)&rhs);
|
||||
}
|
||||
|
||||
void operator=(const double rhs) {
|
||||
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
|
||||
*this = f64_to_extF80(*(float64_t*)&rhs);
|
||||
}
|
||||
|
||||
void operator=(const int16_t rhs) {
|
||||
@@ -493,12 +215,6 @@ struct X80SoftFloat {
|
||||
*this = ui64_to_extF80(rhs);
|
||||
}
|
||||
|
||||
#if BIGFLOATSIZE == 10
|
||||
void operator=(const long double rhs) {
|
||||
memcpy(this, &rhs, sizeof(rhs));
|
||||
}
|
||||
#endif
|
||||
|
||||
operator void*() {
|
||||
return reinterpret_cast<void*>(this);
|
||||
}
|
||||
@@ -510,19 +226,15 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
X80SoftFloat(const float rhs) {
|
||||
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
|
||||
*this = f32_to_extF80(*(float32_t*)&rhs);
|
||||
}
|
||||
|
||||
X80SoftFloat(const double rhs) {
|
||||
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
|
||||
*this = f64_to_extF80(*(float64_t*)&rhs);
|
||||
}
|
||||
|
||||
X80SoftFloat(BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(FEXCore::BitCast<float128_t>(rhs));
|
||||
#else
|
||||
*this = FEXCore::BitCast<long double>(rhs);
|
||||
#endif
|
||||
*this = f128_to_extF80(*(float128_t*)&rhs);
|
||||
}
|
||||
|
||||
X80SoftFloat(const int16_t rhs) {
|
||||
|
||||
-29
@@ -1,29 +0,0 @@
|
||||
#pragma once
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore::StringUtils {
|
||||
// Trim the left side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string LeftTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(TrimTokens)) != std::string::npos) {
|
||||
String.erase(0, pos);
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
// Trim the right side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string RightTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(TrimTokens)) != std::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
return String;
|
||||
}
|
||||
|
||||
// Trim both the left and right of the string of whitespace and new lines
|
||||
[[maybe_unused]] static std::string Trim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
return RightTrim(LeftTrim(String, TrimTokens), TrimTokens);
|
||||
}
|
||||
}
|
||||
+59
-431
@@ -1,127 +1,43 @@
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "Common/Paths.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <cstdlib>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <pwd.h>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <list>
|
||||
#include <optional>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <sys/sysinfo.h>
|
||||
#include <system_error>
|
||||
#include <type_traits>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Config {
|
||||
namespace DefaultValues {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
|
||||
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}
|
||||
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
std::unique_ptr<std::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = std::make_unique<std::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
char const* FindUserHomeThroughUID() {
|
||||
auto passwd = getpwuid(geteuid());
|
||||
if (passwd) {
|
||||
return passwd->pw_dir;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
const char *GetHomeDirectory() {
|
||||
char const *HomeDir = getenv("HOME");
|
||||
|
||||
static void LoadJSonConfig(const std::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
std::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
// Try to get home directory from uid
|
||||
if (!HomeDir) {
|
||||
HomeDir = FindUserHomeThroughUID();
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
|
||||
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
// try the PWD
|
||||
if (!HomeDir) {
|
||||
HomeDir = getenv("PWD");
|
||||
}
|
||||
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
// Still doesn't exit? You get local
|
||||
if (!HomeDir) {
|
||||
HomeDir = ".";
|
||||
}
|
||||
|
||||
for (json_t const* ConfigItem = json_getChild(ConfigList);
|
||||
ConfigItem != nullptr;
|
||||
ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::string GetDataDirectory() {
|
||||
std::string DataDir{};
|
||||
|
||||
char const *HomeDir = Paths::GetHomeDirectory();
|
||||
char const *DataXDG = getenv("XDG_DATA_HOME");
|
||||
char const *DataOverride = getenv("FEX_APP_DATA_LOCATION");
|
||||
if (DataOverride) {
|
||||
// Data override will override the complete directory
|
||||
DataDir = DataOverride;
|
||||
}
|
||||
else {
|
||||
DataDir = DataXDG ?: HomeDir;
|
||||
DataDir += "/.fex-emu/";
|
||||
}
|
||||
return DataDir;
|
||||
return HomeDir;
|
||||
}
|
||||
|
||||
std::string GetConfigDirectory(bool Global) {
|
||||
@@ -130,22 +46,15 @@ namespace JSON {
|
||||
ConfigDir = GLOBAL_DATA_DIRECTORY;
|
||||
}
|
||||
else {
|
||||
char const *HomeDir = Paths::GetHomeDirectory();
|
||||
char const *HomeDir = GetHomeDirectory();
|
||||
char const *ConfigXDG = getenv("XDG_CONFIG_HOME");
|
||||
char const *ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (ConfigOverride) {
|
||||
// Config override completely overrides the config directory
|
||||
ConfigDir = ConfigOverride;
|
||||
}
|
||||
else {
|
||||
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
|
||||
ConfigDir += "/.fex-emu/";
|
||||
}
|
||||
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
|
||||
ConfigDir += "/.fex-emu/";
|
||||
|
||||
// Ensure the folder structure is created for our configuration
|
||||
std::error_code ec{};
|
||||
if (!std::filesystem::exists(ConfigDir, ec) &&
|
||||
!std::filesystem::create_directories(ConfigDir, ec)) {
|
||||
if (!std::filesystem::exists(ConfigDir) &&
|
||||
!std::filesystem::create_directories(ConfigDir)) {
|
||||
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigDir.c_str());
|
||||
// Let's go local in this case
|
||||
return "./";
|
||||
}
|
||||
@@ -154,50 +63,35 @@ namespace JSON {
|
||||
return ConfigDir;
|
||||
}
|
||||
|
||||
std::string GetConfigFileLocation(bool Global) {
|
||||
std::string ConfigFile{};
|
||||
if (Global) {
|
||||
ConfigFile = GetConfigDirectory(true) + "Config.json";
|
||||
}
|
||||
else {
|
||||
const char *AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig) {
|
||||
// App config environment variable overwrites only the config file
|
||||
ConfigFile = AppConfig;
|
||||
}
|
||||
else {
|
||||
ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
}
|
||||
}
|
||||
std::string GetConfigFileLocation() {
|
||||
std::string ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
return ConfigFile;
|
||||
}
|
||||
|
||||
std::string GetApplicationConfig(const std::string &Filename, bool Global) {
|
||||
std::string GetApplicationConfig(std::string &Filename, bool Global) {
|
||||
std::string ConfigFile = GetConfigDirectory(Global);
|
||||
|
||||
std::error_code ec{};
|
||||
if (!Global &&
|
||||
!std::filesystem::exists(ConfigFile, ec) &&
|
||||
!std::filesystem::create_directories(ConfigFile, ec)) {
|
||||
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
|
||||
!std::filesystem::exists(ConfigFile) &&
|
||||
!std::filesystem::create_directories(ConfigFile)) {
|
||||
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigFile.c_str());
|
||||
// Let's go local in this case
|
||||
return "./" + Filename + ".json";
|
||||
return "./";
|
||||
}
|
||||
|
||||
ConfigFile += "AppConfig/";
|
||||
|
||||
// Attempt to create the local folder if it doesn't exist
|
||||
if (!Global &&
|
||||
!std::filesystem::exists(ConfigFile, ec) &&
|
||||
!std::filesystem::create_directories(ConfigFile, ec)) {
|
||||
// Let's go local in this case
|
||||
return "./" + Filename + ".json";
|
||||
}
|
||||
|
||||
ConfigFile += Filename + ".json";
|
||||
ConfigFile += "AppConfig/" + Filename + ".json";
|
||||
return ConfigFile;
|
||||
}
|
||||
|
||||
std::string GetDataDirectory() {
|
||||
std::string DataDir{};
|
||||
|
||||
char const *HomeDir = GetHomeDirectory();
|
||||
char const *DataXDG = getenv("XDG_DATA_HOME");
|
||||
DataDir = DataXDG ?: HomeDir;
|
||||
DataDir += "/.fex-emu/";
|
||||
return DataDir;
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
|
||||
}
|
||||
|
||||
@@ -211,8 +105,7 @@ namespace JSON {
|
||||
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 7> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
|
||||
constexpr std::array<FEXCore::Config::LayerType, 6> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
|
||||
@@ -298,8 +191,7 @@ namespace JSON {
|
||||
void MetaLayer::MergeConfigMap(const LayerOptions &Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto &it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV ||
|
||||
it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV) {
|
||||
MergeEnvironmentVariables(it.first, it.second);
|
||||
}
|
||||
else {
|
||||
@@ -327,7 +219,7 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::string ExpandPath(std::string const &ContainerPrefix, std::string PathName) {
|
||||
std::string ExpandPath(std::string PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
@@ -348,67 +240,10 @@ namespace JSON {
|
||||
Path = std::filesystem::absolute(Path);
|
||||
|
||||
// Only return if it exists
|
||||
std::error_code ec{};
|
||||
if (std::filesystem::exists(Path, ec)) {
|
||||
if (std::filesystem::exists(Path)) {
|
||||
return Path;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// If the containerprefix and pathname isn't empty
|
||||
// Then we check if the pathname exists in our current namespace
|
||||
// If the path DOESN'T exist but DOES exist with the prefix applied
|
||||
// then redirect to the prefix
|
||||
//
|
||||
// This might not be expected behaviour for some edge cases but since
|
||||
// all paths aren't mounted inside the container, then it'll be fine
|
||||
//
|
||||
// Main catch case for this is the default thunk install folders
|
||||
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
|
||||
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
|
||||
if (!ContainerPrefix.empty() && !PathName.empty()) {
|
||||
if (!std::filesystem::exists(PathName)) {
|
||||
auto ContainerPath = ContainerPrefix + PathName;
|
||||
if (std::filesystem::exists(ContainerPath)) {
|
||||
return ContainerPath;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
|
||||
std::string FindContainer() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
return ManagerStr;
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
std::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
|
||||
// We are running inside of pressure vessel
|
||||
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
|
||||
return "/run/host/";
|
||||
}
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
@@ -416,9 +251,7 @@ namespace JSON {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
{
|
||||
// Always fix up the number of threads and create the configuration
|
||||
// Otherwise the application could receive zero as the number of threads
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THREADS)) {
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
if (Cores == 0) {
|
||||
// When the number of emulated CPU cores is zero then auto detect
|
||||
@@ -426,38 +259,8 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
#if (_M_X86_64)
|
||||
constexpr uint32_t MaxCoreNumber = 2;
|
||||
#else
|
||||
constexpr uint32_t MaxCoreNumber = 1;
|
||||
#endif
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
constexpr uint32_t MinCoreNumber = 0;
|
||||
#else
|
||||
constexpr uint32_t MinCoreNumber = 1;
|
||||
#endif
|
||||
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, std::to_string(FEXCore::Config::CONFIG_IRJIT));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
|
||||
if (CacheObjectCodeCompilation() && Core() == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
// If running the interpreter then disable cache code compilation
|
||||
FEXCore::Config::Erase(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION);
|
||||
}
|
||||
}
|
||||
|
||||
std::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
auto ExpandPathIfExists = [](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
auto NewPath = ExpandPath(PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
}
|
||||
@@ -465,7 +268,7 @@ namespace JSON {
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
|
||||
FEX_CONFIG_OPT(PathName, ROOTFS);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
auto ExpandedString = ExpandPath(PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
@@ -473,8 +276,7 @@ namespace JSON {
|
||||
else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
std::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
std::error_code ec{};
|
||||
if (std::filesystem::exists(NamedRootFS, ec)) {
|
||||
if (std::filesystem::exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
@@ -489,19 +291,7 @@ namespace JSON {
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
}
|
||||
else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
std::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
std::error_code ec{};
|
||||
if (std::filesystem::exists(NamedConfig, ec)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKCONFIG, PathName());
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
|
||||
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
|
||||
@@ -532,15 +322,11 @@ namespace JSON {
|
||||
return Meta->Get(Option);
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
void Set(ConfigOption Option, std::string Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
|
||||
void Erase(ConfigOption Option) {
|
||||
Meta->Erase(Option);
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string_view Data) {
|
||||
void EraseSet(ConfigOption Option, std::string Data) {
|
||||
Meta->EraseSet(Option, Data);
|
||||
}
|
||||
|
||||
@@ -579,17 +365,6 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
template<>
|
||||
std::string Value<std::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
}
|
||||
else {
|
||||
return std::string(Default);
|
||||
}
|
||||
}
|
||||
|
||||
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
|
||||
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
|
||||
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
|
||||
@@ -614,152 +389,5 @@ namespace JSON {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<std::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, std::list<std::string> *List);
|
||||
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(std::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
std::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const std::string& Filename, bool Global);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
std::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit EnvLoader(char *const _envp[]);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const std::map<std::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
static const std::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
|
||||
: FEXCore::Config::Layer(Layer) {
|
||||
}
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
auto it = ConfigLookup.find(ConfigName);
|
||||
if (it != ConfigLookup.end()) {
|
||||
Set(it->second, ConfigString);
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(std::string ConfigFile)
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
|
||||
void MainLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const std::string& Filename, bool Global)
|
||||
: FEXCore::Config::OptionMapper(Global ? FEXCore::Config::LayerType::LAYER_GLOBAL_APP : FEXCore::Config::LayerType::LAYER_LOCAL_APP) {
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char *const _envp[])
|
||||
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
|
||||
, envp {_envp} {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
std::unordered_map<std::string_view, std::string_view> EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (std::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
std::string_view Value {Var.substr(pos+1)};
|
||||
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key]=Value;
|
||||
}
|
||||
|
||||
std::function GetVar = [=](const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
// If envp[] was empty, search using std::getenv()
|
||||
const char* vs = std::getenv(id.data());
|
||||
if (vs) {
|
||||
return vs;
|
||||
}
|
||||
else {
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(it.first)).has_value()) {
|
||||
Set(it.second, std::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
|
||||
if (File) {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, bool Global) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Global);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return std::make_unique<FEXCore::Config::EnvLoader>(_envp);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,232 @@
|
||||
{
|
||||
"Options": {
|
||||
"CPU": {
|
||||
"Core": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
|
||||
"TextDefault": "irjit",
|
||||
"ShortArg": "c",
|
||||
"Choices": [ "irint", "irjit", "host" ],
|
||||
"ArgumentHandler": "CoreHandler",
|
||||
"Desc": [
|
||||
"Which CPU core to use",
|
||||
"host only exists on x86_64",
|
||||
"[irint, irjit, host]"
|
||||
]
|
||||
},
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation"
|
||||
]
|
||||
},
|
||||
"MaxInst": {
|
||||
"Type": "int32",
|
||||
"Default": "5000",
|
||||
"ShortArg": "n",
|
||||
"Desc": [
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"Threads": {
|
||||
"Type": "uint32",
|
||||
"Default": "1",
|
||||
"ShortArg": "T",
|
||||
"Desc": [
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
"0 will auto detect."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
"RootFS": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "R",
|
||||
"Desc": [
|
||||
"Which Root filesystem prefix to use",
|
||||
"This can be a filesystem path",
|
||||
"\teg: ~/RootFS/Debian_x86_64",
|
||||
"Or this can be a name of a rootfs",
|
||||
"If the named rootfs exists in the FEX data folder then it will use that one",
|
||||
"\teg: $HOME/.fex-emu/RootFS/<RootFS name>/",
|
||||
"Or if you have XDG_DATA_HOME the config will search in that directory",
|
||||
"\teg: $XDG_DATA_HOME/.fex-emu/RootFS/<RootFS name>/"
|
||||
]
|
||||
},
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "j",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "k",
|
||||
"Desc": [
|
||||
"A json file specifying where to overlay the thunks."
|
||||
]
|
||||
},
|
||||
"Env": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "E",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the emulated environment."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
"SingleStep": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "S",
|
||||
"Desc": [
|
||||
"Single stepping configuration."
|
||||
]
|
||||
},
|
||||
"GdbServer": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "G",
|
||||
"Desc": [
|
||||
"Enables the GDB server."
|
||||
]
|
||||
},
|
||||
"DumpIR": {
|
||||
"Type": "str",
|
||||
"Default": "no",
|
||||
"Desc": [
|
||||
"Folder to dump the IR in to.",
|
||||
"[no, stdout, stderr, <Folder>]"
|
||||
]
|
||||
},
|
||||
"DumpGPRs": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "g",
|
||||
"Desc": [
|
||||
"When the test harness ends, print the GPR state."
|
||||
]
|
||||
},
|
||||
"O0": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "O0",
|
||||
"Desc": [
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Logging": {
|
||||
"SilentLog": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "s",
|
||||
"Desc": [
|
||||
"Disables logging"
|
||||
]
|
||||
},
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "stdout",
|
||||
"ShortArg": "o",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stdout, stderr, <Filename>]"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
"SMCChecks": {
|
||||
"Type": "uint8",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MMAN",
|
||||
"TextDefault": "mman",
|
||||
"ArgumentHandler": "SMCCheckHandler",
|
||||
"Desc": [
|
||||
"Checks code for modification before execution.",
|
||||
"\tnone: No checks",
|
||||
"\tmman: Invalidate on mmap, mprotect, munmap",
|
||||
"\tfull: Validate code before every run (slow)"
|
||||
]
|
||||
},
|
||||
"TSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Controls TSO IR ops.",
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When enabled enables an optimization around flags.",
|
||||
"Assumes flags are not used across cals.",
|
||||
"Hand-written assembly can violate this assumption."
|
||||
]
|
||||
},
|
||||
"ABINoPF": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When enabled enables an optimization around parity flag calculation.",
|
||||
"Removes the calculation of the parity flag from GPR instructions.",
|
||||
"Assuming no uses rely on it"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
"AOTIRCapture": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Captures IR and generates an AOT IR cache.",
|
||||
"Captures both the loaded executable and libraries it loads."
|
||||
]
|
||||
},
|
||||
"AOTIRLoad": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Loads an AOT IR cache for the loaded executable."
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"UnnamedOptions": {
|
||||
"Misc": {
|
||||
"IS_INTERPRETER": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
},
|
||||
"INTERPRETER_INSTALLED": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
},
|
||||
"APP_FILENAME": {
|
||||
"Type": "str",
|
||||
"Default": ""
|
||||
},
|
||||
"IS64BIT_MODE": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,396 +0,0 @@
|
||||
{
|
||||
"Options": {
|
||||
"CPU": {
|
||||
"Core": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
|
||||
"TextDefault": "irjit",
|
||||
"ShortArg": "c",
|
||||
"Choices": [ "irint", "irjit", "host" ],
|
||||
"ArgumentHandler": "CoreHandler",
|
||||
"Desc": [
|
||||
"Which CPU core to use",
|
||||
"host only exists on x86_64",
|
||||
"[irint, irjit, host]"
|
||||
]
|
||||
},
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation"
|
||||
]
|
||||
},
|
||||
"MaxInst": {
|
||||
"Type": "int32",
|
||||
"Default": "5000",
|
||||
"ShortArg": "n",
|
||||
"Desc": [
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"Threads": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"ShortArg": "T",
|
||||
"Desc": [
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
"0 will auto detect."
|
||||
]
|
||||
},
|
||||
"CacheObjectCodeCompilation": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
|
||||
"TextDefault": "none",
|
||||
"Choices": [ "none", "read", "readwrite" ],
|
||||
"ArgumentHandler": "CacheObjectCodeHandler",
|
||||
"Desc": [
|
||||
"Cache JIT object code to drive.",
|
||||
"Allows JIT code to be shared between applications"
|
||||
]
|
||||
},
|
||||
"EnableAVX": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Determines whether or not we use the expanded register file for AVX or not"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
"RootFS": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "R",
|
||||
"Desc": [
|
||||
"Which Root filesystem prefix to use",
|
||||
"This can be a filesystem path",
|
||||
"\teg: ~/RootFS/Debian_x86_64",
|
||||
"Or this can be a name of a rootfs",
|
||||
"If the named rootfs exists in the FEX data folder then it will use that one",
|
||||
"\teg: $HOME/.fex-emu/RootFS/<RootFS name>/",
|
||||
"Or if you have XDG_DATA_HOME the config will search in that directory",
|
||||
"\teg: $XDG_DATA_HOME/.fex-emu/RootFS/<RootFS name>/"
|
||||
]
|
||||
},
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks/",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks/",
|
||||
"ShortArg": "j",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "k",
|
||||
"Desc": [
|
||||
"A json file specifying where to overlay the thunks.",
|
||||
"This can be a filesystem path",
|
||||
"\teg: ~/MyThunkConfig.json",
|
||||
"Or this can be a named of a Thunk config file",
|
||||
"If the named config file exists in the FEX data folder folder the it will use that one",
|
||||
"\teg: $HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>",
|
||||
"Or if you have XDG_DATA_HOME the config will search in that directory",
|
||||
"\teg: $XDG_DATA_HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>"
|
||||
]
|
||||
},
|
||||
"Env": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "E",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the emulated environment."
|
||||
]
|
||||
},
|
||||
"HostEnv": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "H",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the host environment.",
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
"Typically isn't necessary since the guest libc isn't thunked. But is possible."
|
||||
]
|
||||
},
|
||||
"AdditionalArguments": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
"SingleStep": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "S",
|
||||
"Desc": [
|
||||
"Single stepping configuration."
|
||||
]
|
||||
},
|
||||
"GdbServer": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "G",
|
||||
"Desc": [
|
||||
"Enables the GDB server."
|
||||
]
|
||||
},
|
||||
"DumpIR": {
|
||||
"Type": "str",
|
||||
"Default": "no",
|
||||
"Desc": [
|
||||
"Folder to dump the IR in to.",
|
||||
"[no, stdout, stderr, <Folder>]"
|
||||
]
|
||||
},
|
||||
"DumpGPRs": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "g",
|
||||
"Desc": [
|
||||
"When the test harness ends, print the GPR state."
|
||||
]
|
||||
},
|
||||
"O0": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "O0",
|
||||
"Desc": [
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
},
|
||||
"SRA": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Set to false to disable Static Register Allocation"
|
||||
]
|
||||
},
|
||||
"Force32BitAllocator": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Forces use of the 32-bit allocator on 32-bit applications",
|
||||
"Used to work around ulimit problems of CI runner",
|
||||
"Potentially useful for debugging memory problems",
|
||||
"32-bit allocator is always used if your host kernel is older than 4.17"
|
||||
]
|
||||
},
|
||||
"GlobalJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name all JIT state as one symbol",
|
||||
"Useful for querying how much time is spent inside of the JIT",
|
||||
"Profiling tools will show JIT time as FEXJIT"
|
||||
]
|
||||
},
|
||||
"LibraryJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols grouped by library",
|
||||
"Useful for querying how much time is spent in each guest library",
|
||||
"Can be used to help guide thunk generation"
|
||||
]
|
||||
},
|
||||
"BlockJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols",
|
||||
"Useful for determining hot blocks of code",
|
||||
"Has some file writing overhead per JIT block"
|
||||
]
|
||||
},
|
||||
"GDBSymbols": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Integrates with GDB using the JIT interface.",
|
||||
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
|
||||
"Also needs x86_64-linux-gnu-objdump in PATH.",
|
||||
"Can be very slow."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Logging": {
|
||||
"SilentLog": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "s",
|
||||
"Desc": [
|
||||
"Disables logging"
|
||||
]
|
||||
},
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "server",
|
||||
"ShortArg": "o",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stdout, stderr, server, <Filename>]"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
"SMCChecks": {
|
||||
"Type": "uint8",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
|
||||
"TextDefault": "mtrack",
|
||||
"ArgumentHandler": "SMCCheckHandler",
|
||||
"Desc": [
|
||||
"Checks code for modification before execution.",
|
||||
"\tnone: No checks",
|
||||
"\tmtrack: Page tracking based invalidation",
|
||||
"\tfull: Validate code before every run (slow)",
|
||||
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
|
||||
]
|
||||
},
|
||||
"TSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Controls TSO IR ops.",
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Automatically enables TSO when shared memory is used.",
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When enabled enables an optimization around flags.",
|
||||
"Assumes flags are not used across cals.",
|
||||
"Hand-written assembly can violate this assumption."
|
||||
]
|
||||
},
|
||||
"ABINoPF": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When enabled enables an optimization around parity flag calculation.",
|
||||
"Removes the calculation of the parity flag from GPR instructions.",
|
||||
"Assuming no uses rely on it"
|
||||
]
|
||||
},
|
||||
"ParanoidTSO": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Makes TSO operations even more strict.",
|
||||
"Forces vector loadstores to also become atomic."
|
||||
]
|
||||
},
|
||||
"StallProcess": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Forces a process to stall out on initialization",
|
||||
"Useful for a process that keeps restarting and doesn't work"
|
||||
]
|
||||
},
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"An application that uses try-catch or longjump extensively needs the ability to do context aware state flushing",
|
||||
"In the case of FEX's block-linking, it won't always ensure that RIP is synchronized.",
|
||||
"If an exception occurs and RIP isn't synchronized, then FEX's exception stack restore may not long jump as expected",
|
||||
"Can be useful for Wine applications that rely on stack unwinding"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
"AOTIRCapture": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Captures IR and generates an AOT IR cache.",
|
||||
"Captures both the loaded executable and libraries it loads."
|
||||
]
|
||||
},
|
||||
"AOTIRGenerate": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Scans file for executable code and generates an AOT IR cache.",
|
||||
"Does not run the executable."
|
||||
]
|
||||
},
|
||||
"AOTIRLoad": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Loads an AOT IR cache for the loaded executable."
|
||||
]
|
||||
},
|
||||
"ServerSocketPath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Override for a FEXServer socket path. Only useful for chroots."
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"UnnamedOptions": {
|
||||
"Misc": {
|
||||
"IS_INTERPRETER": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
},
|
||||
"INTERPRETER_INSTALLED": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
},
|
||||
"APP_FILENAME": {
|
||||
"Type": "str",
|
||||
"Default": ""
|
||||
},
|
||||
"APP_CONFIG_NAME": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"This is the application config name that has been loaded.",
|
||||
"This differs from APP_FILENAME in two ways",
|
||||
"Where APP_FILENAME always points to the executable path that FEX-Emu is executing.",
|
||||
"This matches what is used to load the AppLayer configuration name.",
|
||||
"When running through a compatibility layer like wine, this will only be the exe name, instead of wine full path."
|
||||
]
|
||||
},
|
||||
"IS64BIT_MODE": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+28
-77
@@ -2,20 +2,10 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <string.h>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::HLE {
|
||||
class SyscallVisitor;
|
||||
}
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void InitializeStaticTables(OperatingMode Mode) {
|
||||
@@ -24,10 +14,6 @@ namespace FEXCore::Context {
|
||||
IR::InstallOpcodeHandlers(Mode);
|
||||
}
|
||||
|
||||
void ShutdownStaticTables() {
|
||||
FEXCore::Paths::ShutdownPaths();
|
||||
}
|
||||
|
||||
FEXCore::Context::Context *CreateNewContext() {
|
||||
return new FEXCore::Context::Context{};
|
||||
}
|
||||
@@ -43,15 +29,16 @@ namespace FEXCore::Context {
|
||||
delete CTX;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
return CTX->InitCore(InitialRIP, StackPointer);
|
||||
bool InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
|
||||
return CTX->InitCore(Loader);
|
||||
}
|
||||
|
||||
void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler) {
|
||||
CTX->CustomExitHandler = std::move(handler);
|
||||
void SetExitHandler(FEXCore::Context::Context *CTX,
|
||||
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> handler) {
|
||||
CTX->CustomExitHandler = handler;
|
||||
}
|
||||
|
||||
ExitHandler GetExitHandler(const FEXCore::Context::Context *CTX) {
|
||||
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> GetExitHandler(FEXCore::Context::Context *CTX) {
|
||||
return CTX->CustomExitHandler;
|
||||
}
|
||||
|
||||
@@ -63,31 +50,28 @@ namespace FEXCore::Context {
|
||||
CTX->Step();
|
||||
}
|
||||
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
Thread->CTX->CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason RunUntilExit(FEXCore::Context::Context *CTX) {
|
||||
return CTX->RunUntilExit();
|
||||
}
|
||||
|
||||
int GetProgramStatus(const FEXCore::Context::Context *CTX) {
|
||||
int GetProgramStatus(FEXCore::Context::Context *CTX) {
|
||||
return CTX->GetProgramStatus();
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason GetExitReason(const FEXCore::Context::Context *CTX) {
|
||||
FEXCore::Context::ExitReason GetExitReason(FEXCore::Context::Context *CTX) {
|
||||
return CTX->ParentThread->ExitReason;
|
||||
}
|
||||
|
||||
bool IsDone(const FEXCore::Context::Context *CTX) {
|
||||
bool IsDone(FEXCore::Context::Context *CTX) {
|
||||
return CTX->IsPaused();
|
||||
}
|
||||
|
||||
void GetCPUState(const FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
void GetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
memcpy(State, CTX->ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
void SetCPUState(FEXCore::Context::Context *CTX, const FEXCore::Core::CPUState *State) {
|
||||
void SetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
memcpy(CTX->ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
@@ -110,30 +94,22 @@ namespace FEXCore::Context {
|
||||
void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t Syscall, [[maybe_unused]] FEXCore::HLE::SyscallVisitor *Visitor) {
|
||||
}
|
||||
|
||||
HostFeatures GetHostFeatures(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->HostFeatures;
|
||||
void HandleCallback(FEXCore::Context::Context *CTX, uint64_t RIP) {
|
||||
CTX->HandleCallback(RIP);
|
||||
}
|
||||
|
||||
void HandleCallback(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
CTX->HandleCallback(Thread, RIP);
|
||||
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
|
||||
CTX->RegisterHostSignalHandler(Signal, Func);
|
||||
}
|
||||
|
||||
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterHostSignalHandler(Signal, std::move(Func), Required);
|
||||
}
|
||||
|
||||
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterFrontendHostSignalHandler(Signal, std::move(Func), Required);
|
||||
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
|
||||
CTX->RegisterFrontendHostSignalHandler(Signal, Func);
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
return CTX->CreateThread(NewThreadState, ParentTID);
|
||||
}
|
||||
|
||||
void ExecutionThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return CTX->ExecutionThread(Thread);
|
||||
}
|
||||
|
||||
void InitializeThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return CTX->InitializeThread(Thread);
|
||||
}
|
||||
@@ -153,57 +129,32 @@ namespace FEXCore::Context {
|
||||
void CleanupAfterFork(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
CTX->CleanupAfterFork(Thread);
|
||||
}
|
||||
|
||||
|
||||
void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation) {
|
||||
CTX->SignalDelegation = SignalDelegation;
|
||||
}
|
||||
|
||||
void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler) {
|
||||
CTX->SyscallHandler = Handler;
|
||||
CTX->SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf) {
|
||||
return CTX->CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CTX->CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::istream>(const std::string&)> CacheReader) {
|
||||
CTX->AOTIRLoader = CacheReader;
|
||||
}
|
||||
|
||||
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader) {
|
||||
CTX->SetAOTIRLoader(CacheReader);
|
||||
bool WriteAOTIR(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
|
||||
return CTX->WriteAOTIRCache(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
|
||||
CTX->SetAOTIRWriter(CacheWriter);
|
||||
void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name) {
|
||||
return CTX->AddNamedRegion(Base, Length, Offset, Name);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(FEXCore::Context::Context *CTX, std::function<void(const std::string&)> CacheRenamer) {
|
||||
CTX->SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache(FEXCore::Context::Context *CTX) {
|
||||
CTX->FinalizeAOTIRCache();
|
||||
}
|
||||
|
||||
void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
CTX->WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(FEXCore::Context::Context *CTX, const std::string &Name) {
|
||||
return CTX->LoadAOTIRCacheEntry(Name);
|
||||
}
|
||||
void UnloadAOTIRCacheEntry(FEXCore::Context::Context *CTX, IR::AOTIRCacheEntry *Entry) {
|
||||
return CTX->UnloadAOTIRCacheEntry(Entry);
|
||||
}
|
||||
|
||||
CustomIRResult AddCustomIREntrypoint(FEXCore::Context::Context *CTX, uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
return CTX->AddCustomIREntrypoint(Entrypoint, Handler, Creator, Data);
|
||||
}
|
||||
|
||||
void AppendThunkDefinitions(FEXCore::Context::Context *CTX, std::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
CTX->AppendThunkDefinitions(Definitions);
|
||||
void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length) {
|
||||
return CTX->RemoveNamedRegion(Base, Length);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
|
||||
+76
-213
@@ -1,61 +1,44 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "FEXHeaderUtils/ScopedSignalMask.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <stdint.h>
|
||||
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <functional>
|
||||
#include <istream>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <stddef.h>
|
||||
#include <string>
|
||||
#include <optional>
|
||||
#include <ostream>
|
||||
#include <set>
|
||||
#include <unordered_map>
|
||||
#include <queue>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
class ThunkHandler;
|
||||
class BlockSamplingData;
|
||||
class GdbServer;
|
||||
|
||||
namespace CodeSerialize {
|
||||
class CodeObjectSerializeService;
|
||||
}
|
||||
class SiganlDelegator;
|
||||
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class X86JITCore;
|
||||
class InterpreterCore;
|
||||
class Dispatcher;
|
||||
}
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
struct SourcecodeMap;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
namespace Validation {
|
||||
@@ -78,7 +61,6 @@ namespace FEXCore::Context {
|
||||
friend class FEXCore::CPU::X86JITCore;
|
||||
#endif
|
||||
|
||||
friend class FEXCore::CPU::InterpreterCore;
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
@@ -93,38 +75,28 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(ABINoPF, ABINOPF);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
|
||||
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
|
||||
FEX_CONFIG_OPT(DumpIR, DUMPIR);
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(x86dec_SynchronizeRIPOnAllBlocks, X86DEC_SYNCHRONIZERIPONALLBLOCKS);
|
||||
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
|
||||
} Config;
|
||||
|
||||
using IntCallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
IntCallbackReturn InterpreterCallbackReturn;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
|
||||
std::mutex ThreadCreationMutex;
|
||||
uint64_t ThreadID{};
|
||||
FEXCore::Core::InternalThreadState* ParentThread;
|
||||
std::vector<FEXCore::Core::InternalThreadState*> Threads;
|
||||
std::atomic_bool CoreShuttingDown{false};
|
||||
bool NeedToCheckXID{true};
|
||||
|
||||
std::mutex IdleWaitMutex;
|
||||
std::condition_variable IdleWaitCV;
|
||||
@@ -133,16 +105,34 @@ namespace FEXCore::Context {
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
std::shared_mutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
|
||||
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> CustomExitHandler;
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
uint64_t start;
|
||||
uint64_t len;
|
||||
uint64_t crc;
|
||||
IR::IRListView *IR;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
};
|
||||
|
||||
std::function<std::unique_ptr<std::istream>(const std::string&)> AOTIRLoader;
|
||||
std::unordered_map<std::string, std::map<uint64_t, AOTIRCacheEntry>> AOTIRCache;
|
||||
|
||||
struct AddrToFileEntry {
|
||||
uint64_t Start;
|
||||
uint64_t Len;
|
||||
uint64_t Offset;
|
||||
std::string fileid;
|
||||
void *CachedFileEntry;
|
||||
};
|
||||
|
||||
std::map<uint64_t, AddrToFileEntry> AddrToFile;
|
||||
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
@@ -154,9 +144,9 @@ namespace FEXCore::Context {
|
||||
Context();
|
||||
~Context();
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer);
|
||||
bool InitCore(FEXCore::CodeLoader *Loader);
|
||||
FEXCore::Context::ExitReason RunUntilExit();
|
||||
int GetProgramStatus() const;
|
||||
int GetProgramStatus();
|
||||
bool IsPaused() const { return !Running; }
|
||||
void Pause();
|
||||
void Run();
|
||||
@@ -167,40 +157,20 @@ namespace FEXCore::Context {
|
||||
void StopThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
|
||||
|
||||
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
|
||||
bool GetGdbServerStatus() { return (bool)DebugServer; }
|
||||
void StartGdbServer();
|
||||
void StopGdbServer();
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void HandleCallback(uint64_t RIP);
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
|
||||
static void RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
FHU::ScopedSignalMaskWithSharedLock lk(Frame->Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
return Fn(Frame, record);
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState
|
||||
static void RemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
RemoveCodeEntry(Frame->Thread, GuestRIP);
|
||||
}
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data);
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
// Debugger interface
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
uint64_t GetThreadCount() const;
|
||||
@@ -208,172 +178,65 @@ namespace FEXCore::Context {
|
||||
bool GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data);
|
||||
bool FindHostCodeForRIP(uint64_t RIP, uint8_t **Code);
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
|
||||
// XXX:
|
||||
// bool FindIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir);
|
||||
// void SetIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir);
|
||||
void LoadEntryList();
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
FEXCore::IR::IRListView* IRData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
bool GeneratedIR;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
std::tuple<FEXCore::IR::IRListView *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t, uint64_t, uint64_t> GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
std::tuple<void *, FEXCore::IR::IRListView *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool, uint64_t, uint64_t> CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
bool LoadAOTIRCache(std::istream &stream);
|
||||
bool WriteAOTIRCache(std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - InitializeThread(Thread);
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
|
||||
|
||||
/**
|
||||
* @brief Initializes TID, PID and TLS data for a thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* The OS thread will wait until RunThread is executed
|
||||
*/
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Starts the OS thread object to start executing guest code
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
void RunThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
|
||||
void CleanupAfterFork(FEXCore::Core::InternalThreadState *ExceptForThread);
|
||||
|
||||
std::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
std::vector<FEXCore::Core::InternalThreadState*> *const GetThreads() { return &Threads; }
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const std::string &filename);
|
||||
void UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry);
|
||||
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
|
||||
|
||||
#if ENABLE_JITSYMBOLS
|
||||
FEXCore::JITSymbols Symbols;
|
||||
#endif
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void FinalizeAOTIRCache() {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
|
||||
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void SetAOTIRLoader(std::function<int(const std::string&)> CacheReader) {
|
||||
IRCaptureCache.SetAOTIRLoader(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
|
||||
IRCaptureCache.SetAOTIRWriter(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(std::function<void(const std::string&)> CacheRenamer) {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void AppendThunkDefinitions(std::vector<FEXCore::IR::ThunkDefinition> const& Definitions);
|
||||
|
||||
FEXCore::Utils::PooledAllocatorMMap OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorMMap FrontendAllocator;
|
||||
|
||||
void MarkMemoryShared();
|
||||
|
||||
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Does some final thread initialization
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* InitCore and CreateThread both call this to finish up thread object initialization
|
||||
*/
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void WaitForIdleWithTimeout();
|
||||
|
||||
void NotifyPause();
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length);
|
||||
|
||||
FEXCore::CodeLoader *LocalLoader{};
|
||||
|
||||
// Entry Cache
|
||||
std::optional<std::string> GetFilenameHash(std::string const &Filename) const;
|
||||
void AddThreadRIPsToEntryList(FEXCore::Core::InternalThreadState *Thread);
|
||||
void SaveEntryList();
|
||||
std::set<uint64_t> EntryList;
|
||||
std::vector<uint64_t> InitLocations;
|
||||
uint64_t StartingRIP;
|
||||
std::mutex ExitMutex;
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
std::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
|
||||
|
||||
+215
-1388
File diff suppressed because it is too large.
Load diff
@@ -12,58 +12,6 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
|
||||
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
|
||||
|
||||
constexpr uint32_t RCPC2_MASK = 0x3F'E0'0C'00;
|
||||
constexpr uint32_t LDAPUR_INST = 0x19'40'00'00;
|
||||
constexpr uint32_t STLUR_INST = 0x19'00'00'00;
|
||||
|
||||
constexpr uint32_t LDAXP_MASK = 0xBF'FF'80'00;
|
||||
constexpr uint32_t LDAXP_INST = 0x88'7F'80'00;
|
||||
|
||||
constexpr uint32_t STLXP_MASK = 0xBF'E0'80'00;
|
||||
constexpr uint32_t STLXP_INST = 0x88'20'80'00;
|
||||
|
||||
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
|
||||
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
|
||||
|
||||
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
|
||||
|
||||
constexpr uint32_t CBNZ_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t CBNZ_INST = 0x35'00'00'00;
|
||||
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'20'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t ADD_SHIFT_INST = 0x0B'20'00'00;
|
||||
constexpr uint32_t SUB_SHIFT_INST = 0x4B'20'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t CMP_SHIFT_INST = 0x6B'20'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t BIC_INST = 0x0A'20'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t ORN_INST = 0x2A'20'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
constexpr uint32_t EON_INST = 0x4A'20'00'00;
|
||||
|
||||
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
|
||||
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
|
||||
|
||||
constexpr uint32_t CLREX_MASK = 0xFF'FF'F0'FF;
|
||||
constexpr uint32_t CLREX_INST = 0xD5'03'30'5F;
|
||||
|
||||
enum ExclusiveAtomicPairType {
|
||||
TYPE_SWAP,
|
||||
TYPE_ADD,
|
||||
TYPE_SUB,
|
||||
TYPE_AND,
|
||||
TYPE_BIC,
|
||||
TYPE_OR,
|
||||
TYPE_ORN,
|
||||
TYPE_EOR,
|
||||
TYPE_EON,
|
||||
TYPE_NEG, // This is just a sub with zero. Need to know the differences
|
||||
};
|
||||
|
||||
// Load ops are 4 bits
|
||||
// Acquire and release bits are independent on the instruction
|
||||
constexpr uint32_t ATOMIC_ADD_OP = 0b0000;
|
||||
@@ -76,34 +24,7 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t ATOMIC_UMIN_OP = 0b0111;
|
||||
constexpr uint32_t ATOMIC_SWAP_OP = 0b1000;
|
||||
|
||||
constexpr uint32_t REGISTER_MASK = 0b11111;
|
||||
constexpr uint32_t RD_OFFSET = 0;
|
||||
constexpr uint32_t RN_OFFSET = 5;
|
||||
constexpr uint32_t RM_OFFSET = 16;
|
||||
|
||||
constexpr uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
|
||||
0b1011'0000'0000; // Inner shareable all
|
||||
|
||||
inline uint32_t GetRdReg(uint32_t Instr) {
|
||||
return (Instr >> RD_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
inline uint32_t GetRnReg(uint32_t Instr) {
|
||||
return (Instr >> RN_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
inline uint32_t GetRmReg(uint32_t Instr) {
|
||||
return (Instr >> RM_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr);
|
||||
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info);
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr);
|
||||
uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr);
|
||||
[[nodiscard]] bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext);
|
||||
}
|
||||
+48
-169
@@ -1,110 +1,50 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/instructions-aarch64.h>
|
||||
#include <cpu-features.h>
|
||||
#include <utils-vixl.h>
|
||||
|
||||
#include <array>
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
: vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode)
|
||||
, EmitterCTX {ctx} {
|
||||
Arm64Emitter::Arm64Emitter(size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
|
||||
CPU.SetUp();
|
||||
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
if (ctx->HostFeatures.SupportsAtomics) {
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
// RCPC is bugged on Snapdragon 865
|
||||
// Causes glibc cond16 test to immediately throw assert
|
||||
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
|
||||
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
|
||||
if (SupportsAtomics) {
|
||||
// Hypervisor can hide this on the c630?
|
||||
Features.Combine(vixl::CPUFeatures::Feature::kLORegions);
|
||||
}
|
||||
|
||||
SetCPUFeatures(Features);
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
bool Is64Bit = Reg.IsX();
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
|
||||
if (Is64Bit && ((~Constant)>> 16) == 0) {
|
||||
movn(Reg, (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
int NumMoves = 1;
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
// Count the number of move segments
|
||||
// We only want to use ADRP+ADD if we have more than 1 segment
|
||||
for (size_t i = 0; i < Segments; ++i) {
|
||||
movz(Reg, (Constant) & 0xFFFF, 0);
|
||||
for (int i = 1; i < Segments; ++i) {
|
||||
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
|
||||
if (Part != 0) {
|
||||
++RequiredMoveSegments;
|
||||
}
|
||||
}
|
||||
|
||||
// ADRP+ADD is specifically optimized in hardware
|
||||
// Check if we can use this
|
||||
auto PC = GetCursorAddress<uint64_t>();
|
||||
|
||||
// PC aligned to page
|
||||
uint64_t AlignedPC = PC & ~0xFFFULL;
|
||||
|
||||
// Offset from aligned PC
|
||||
int64_t AlignedOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(AlignedPC);
|
||||
|
||||
// If the aligned offset is within the 4GB window then we can use ADRP+ADD
|
||||
// and the number of move segments more than 1
|
||||
if (RequiredMoveSegments > 1 && vixl::IsInt32(AlignedOffset)) {
|
||||
// If this is 4k page aligned then we only need ADRP
|
||||
if ((AlignedOffset & 0xFFF) == 0) {
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
}
|
||||
else {
|
||||
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
|
||||
// 21-bit signed integer here
|
||||
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
|
||||
if (vixl::IsInt21(SmallOffset)) {
|
||||
adr(Reg, SmallOffset);
|
||||
}
|
||||
else {
|
||||
// Need to use ADRP + ADD
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
add(Reg, Reg, Constant & 0xFFF);
|
||||
NumMoves = 2;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
movz(Reg, (Constant) & 0xFFFF, 0);
|
||||
for (int i = 1; i < Segments; ++i) {
|
||||
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
|
||||
if (Part) {
|
||||
movk(Reg, Part, i * 16);
|
||||
++NumMoves;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (NOPPad) {
|
||||
for (int i = NumMoves; i < Segments; ++i) {
|
||||
nop();
|
||||
if (Part) {
|
||||
movk(Reg, Part, i * 16);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -113,7 +53,7 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We need to save pairs of registers
|
||||
// We save r19-r30
|
||||
MemOperand PairOffset(sp, -16, PreIndex);
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
|
||||
{x19, x20},
|
||||
{x21, x22},
|
||||
{x23, x24},
|
||||
@@ -176,7 +116,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
MemOperand PairOffset(sp, 16, PostIndex);
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
|
||||
{x29, x30},
|
||||
{x27, x28},
|
||||
{x25, x26},
|
||||
@@ -191,97 +131,23 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
if (StaticRegisterAllocation()) {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.GetCode()) & GPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & GPRSpillMask)) {
|
||||
stp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & GPRSpillMask)) {
|
||||
str(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & GPRSpillMask)) {
|
||||
str(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
|
||||
}
|
||||
}
|
||||
void Arm64Emitter::SpillStaticRegs() {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
stp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRSpillMask) != 0) {
|
||||
str(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
stp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
stp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
if (StaticRegisterAllocation()) {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.GetCode()) & GPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & GPRFillMask)) {
|
||||
ldp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & GPRFillMask)) {
|
||||
ldr(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & GPRFillMask)) {
|
||||
ldr(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
|
||||
}
|
||||
}
|
||||
void Arm64Emitter::FillStaticRegs() {
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
ldp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRFillMask) != 0) {
|
||||
ldr(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
ldp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -331,10 +197,23 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
add(sp, sp, SPOffset);
|
||||
}
|
||||
|
||||
void Arm64Emitter::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
return;
|
||||
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::Align16B() {
|
||||
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
|
||||
uint64_t CurrentOffset = GetBuffer()->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,20 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
@@ -62,19 +49,15 @@ const std::array<aarch64::VRegister, 12> RAFPR = {
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
Arm64Emitter(size_t size);
|
||||
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
bool SupportsAtomics{};
|
||||
bool SupportsRCPC{};
|
||||
|
||||
static constexpr uint32_t CALLER_GPR_MASK = 0b0011'1111'1111'1111'1111;
|
||||
|
||||
// This isn't technically true because the lower 64-bits of v8..v15 are callee saved
|
||||
// We can't guarantee only the lower 64bits are used so flush everything
|
||||
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
void SpillStaticRegs();
|
||||
void FillStaticRegs();
|
||||
|
||||
void PushDynamicRegsAndLR();
|
||||
void PopDynamicRegsAndLR();
|
||||
@@ -82,9 +65,10 @@ protected:
|
||||
void PushCalleeSavedRegisters();
|
||||
void PopCalleeSavedRegisters();
|
||||
|
||||
void ResetStack();
|
||||
void Align16B();
|
||||
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
uint32_t SpillSlots{};
|
||||
};
|
||||
|
||||
}
|
||||
@@ -1,7 +1,6 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
|
||||
@@ -12,16 +11,16 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
// Obvously such a configuration can't do the actual arm64-specific stuff
|
||||
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
ERROR_AND_DIE_FMT("HandleCASPAL Not Implemented");
|
||||
ERROR_AND_DIE("HandleCASPAL Not Implemented");
|
||||
}
|
||||
|
||||
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
ERROR_AND_DIE_FMT("HandleCASAL Not Implemented");
|
||||
ERROR_AND_DIE("HandleCASAL Not Implemented");
|
||||
}
|
||||
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
|
||||
ERROR_AND_DIE("HandleAtomicMemOp Not Implemented");
|
||||
}
|
||||
#endif
|
||||
|
||||
}
|
||||
}
|
||||
+22
-77
@@ -13,27 +13,16 @@
|
||||
|
||||
namespace FEXCore::ArchHelpers::Context {
|
||||
|
||||
enum ContextFlags : uint32_t {
|
||||
CONTEXT_FLAG_INJIT = (1U << 0),
|
||||
CONTEXT_FLAG_32BIT = (1U << 1),
|
||||
};
|
||||
|
||||
struct X86ContextBackup {
|
||||
// Host State
|
||||
// RIP and RSP is stored in GPRs here
|
||||
uint64_t GPRs[23];
|
||||
FEXCore::x86_64::_libc_fpstate FPRState;
|
||||
uint64_t sa_mask;
|
||||
bool FaultToTopAndGeneratedException;
|
||||
|
||||
// Guest state
|
||||
int Signal;
|
||||
uint32_t Flags;
|
||||
uint64_t OriginalRIP;
|
||||
uint64_t FPStateLocation;
|
||||
uint64_t UContextLocation;
|
||||
uint64_t SigInfoLocation;
|
||||
FEXCore::Core::CPUState GuestState;
|
||||
|
||||
static constexpr int RedZoneSize = 128;
|
||||
};
|
||||
|
||||
@@ -46,27 +35,15 @@ struct ArmContextBackup {
|
||||
uint32_t FPSR;
|
||||
uint32_t FPCR;
|
||||
__uint128_t FPRs[32];
|
||||
uint64_t sa_mask;
|
||||
bool FaultToTopAndGeneratedException;
|
||||
|
||||
// Guest state
|
||||
int Signal;
|
||||
uint32_t Flags;
|
||||
uint64_t OriginalRIP;
|
||||
uint64_t FPStateLocation;
|
||||
uint64_t UContextLocation;
|
||||
uint64_t SigInfoLocation;
|
||||
FEXCore::Core::CPUState GuestState;
|
||||
|
||||
// Arm64 doesn't have a red zone
|
||||
static constexpr int RedZoneSize = 0;
|
||||
};
|
||||
|
||||
static inline ucontext_t* GetUContext(void* ucontext) {
|
||||
ucontext_t* _context = (ucontext_t*)ucontext;
|
||||
return _context;
|
||||
}
|
||||
|
||||
static inline mcontext_t* GetMContext(void* ucontext) {
|
||||
ucontext_t* _context = (ucontext_t*)ucontext;
|
||||
return &_context->uc_mcontext;
|
||||
@@ -75,20 +52,6 @@ static inline mcontext_t* GetMContext(void* ucontext) {
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
|
||||
constexpr uint32_t FPR_MAGIC = 0x46508001U;
|
||||
|
||||
struct HostCTXHeader {
|
||||
uint32_t Magic;
|
||||
uint32_t Size;
|
||||
};
|
||||
|
||||
struct HostFPRState {
|
||||
HostCTXHeader Head;
|
||||
uint32_t FPSR;
|
||||
uint32_t FPCR;
|
||||
__uint128_t FPRs[32];
|
||||
};
|
||||
|
||||
static inline uint64_t GetSp(void* ucontext) {
|
||||
return GetMContext(ucontext)->sp;
|
||||
}
|
||||
@@ -121,19 +84,24 @@ static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
|
||||
GetMContext(ucontext)->regs[id] = val;
|
||||
}
|
||||
|
||||
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
auto MContext = GetMContext(ucontext);
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&MContext->__reserved[0]);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
constexpr uint32_t FPR_MAGIC = 0x46508001U;
|
||||
|
||||
return HostState->FPRs[id];
|
||||
}
|
||||
struct HostCTXHeader {
|
||||
uint32_t Magic;
|
||||
uint32_t Size;
|
||||
};
|
||||
|
||||
struct HostFPRState {
|
||||
HostCTXHeader Head;
|
||||
uint32_t FPSR;
|
||||
uint32_t FPCR;
|
||||
__uint128_t FPRs[32];
|
||||
};
|
||||
|
||||
using ContextBackup = ArmContextBackup;
|
||||
template <typename T>
|
||||
static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, ArmContextBackup>::value) {
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
memcpy(&Backup->GPRs[0], &_mcontext->regs[0], 31 * sizeof(uint64_t));
|
||||
@@ -143,27 +111,22 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
|
||||
// Host FPR state starts at _mcontext->reserved[0];
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LogMan::Throw::A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
|
||||
Backup->FPSR = HostState->FPSR;
|
||||
Backup->FPCR = HostState->FPCR;
|
||||
memcpy(&Backup->FPRs[0], &HostState->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
|
||||
// Save the signal mask so we can restore it
|
||||
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
|
||||
} else {
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, ArmContextBackup>::value) {
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LogMan::Throw::A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
|
||||
memcpy(&HostState->FPRs[0], &Backup->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
HostState->FPCR = Backup->FPCR;
|
||||
HostState->FPSR = Backup->FPSR;
|
||||
@@ -173,12 +136,8 @@ static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
ArchHelpers::Context::SetPc(ucontext, Backup->PrevPC);
|
||||
ArchHelpers::Context::SetSp(ucontext, Backup->PrevSP);
|
||||
memcpy(&_mcontext->regs[0], &Backup->GPRs[0], 31 * sizeof(uint64_t));
|
||||
|
||||
// Restore the signal mask now
|
||||
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
|
||||
} else {
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
|
||||
}
|
||||
}
|
||||
|
||||
@@ -211,22 +170,17 @@ static inline void SetState(void* ucontext, uint64_t val) {
|
||||
}
|
||||
|
||||
static inline uint64_t GetArmReg(void* ucontext, uint32_t id) {
|
||||
ERROR_AND_DIE_FMT("Not impelented for x86 host");
|
||||
ERROR_AND_DIE("Not impelented for x86 host");
|
||||
}
|
||||
|
||||
static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
|
||||
ERROR_AND_DIE_FMT("Not impelented for x86 host");
|
||||
}
|
||||
|
||||
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
ERROR_AND_DIE("Not impelented for x86 host");
|
||||
}
|
||||
|
||||
using ContextBackup = X86ContextBackup;
|
||||
template <typename T>
|
||||
static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, X86ContextBackup>::value) {
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
// Copy the GPRs
|
||||
@@ -234,34 +188,25 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
// Copy the FPRState
|
||||
memcpy(&Backup->FPRState, _mcontext->fpregs, sizeof(X86ContextBackup::FPRState));
|
||||
// XXX: Save 256bit and 512bit AVX register state
|
||||
|
||||
// Save the signal mask so we can restore it
|
||||
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
|
||||
} else {
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, X86ContextBackup>::value) {
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
// Copy the GPRs
|
||||
memcpy(&_mcontext->gregs[0], &Backup->GPRs[0], sizeof(X86ContextBackup::GPRs));
|
||||
// Copy the FPRState
|
||||
memcpy(_mcontext->fpregs, &Backup->FPRState, sizeof(X86ContextBackup::FPRState));
|
||||
|
||||
// Restore the signal mask now
|
||||
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
|
||||
} else {
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
} // namespace FEXCore::ArchHelpers::Context
|
||||
} // namespace FEXCore::ArchHelpers::Context
|
||||
@@ -2,7 +2,6 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore {
|
||||
void BlockSamplingData::DumpBlockData() {
|
||||
@@ -27,7 +26,7 @@ namespace FEXCore {
|
||||
<< std::endl;
|
||||
}
|
||||
Output.close();
|
||||
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
|
||||
LogMan::Msg::D("Dumped %d blocks of sampling data", SamplingMap.size());
|
||||
}
|
||||
|
||||
BlockSamplingData::BlockData *BlockSamplingData::GetBlockData(uint64_t RIP) {
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <unordered_map>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore {
|
||||
class BlockSamplingData {
|
||||
|
||||
@@ -1,87 +0,0 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {}
|
||||
|
||||
CPUBackend::~CPUBackend() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
}
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
|
||||
if (CodeBuffers.empty()) {
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
} else {
|
||||
if (CodeBuffers.size() > 1) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (size_t i = 1; i < CodeBuffers.size(); i++) {
|
||||
FreeCodeBuffer(CodeBuffers[i]);
|
||||
}
|
||||
CodeBuffers.resize(1);
|
||||
}
|
||||
// Set the current code buffer to the initial
|
||||
CurrentCodeBuffer = &CodeBuffers[0];
|
||||
|
||||
if (CurrentCodeBuffer->Size != MaxCodeSize) {
|
||||
FreeCodeBuffer(*CurrentCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
|
||||
|
||||
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
}
|
||||
|
||||
return CurrentCodeBuffer;
|
||||
}
|
||||
|
||||
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t *>(
|
||||
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
if (ThreadState->CTX->Config.GlobalJITNaming()) {
|
||||
ThreadState->CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
for (auto &Buffer: CodeBuffers) {
|
||||
auto start = (uintptr_t)Buffer.Ptr;
|
||||
auto end = start + Buffer.Size;
|
||||
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
+174
-911
File diff suppressed because it is too large.
Load diff
+26
-66
@@ -1,12 +1,9 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
@@ -24,83 +21,46 @@ private:
|
||||
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
|
||||
|
||||
public:
|
||||
// X86 cacheline size effectively has to be hardcoded to 64
|
||||
// if we report anything differently then applications are likely to break
|
||||
constexpr static uint64_t CACHELINE_SIZE = 64;
|
||||
|
||||
void Init(FEXCore::Context::Context *ctx);
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) {
|
||||
const auto Handler = FunctionHandlers.find(Function);
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, [[maybe_unused]] uint32_t Leaf) {
|
||||
auto Handler = FunctionHandlers.find(Function);
|
||||
|
||||
if (Handler == FunctionHandlers.end()) {
|
||||
return Function_Reserved(Leaf);
|
||||
#ifndef NDEBUG
|
||||
LogMan::Msg::E("Unhandled CPU ID function, 0x%x", Function);
|
||||
#endif
|
||||
return Function_Reserved();
|
||||
}
|
||||
|
||||
return (this->*Handler->second)(Leaf);
|
||||
return Handler->second();
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
if (Function == 0x8000'0002U)
|
||||
return Function_8000_0002h(Leaf, CPU % PerCPUData.size());
|
||||
else if (Function == 0x8000'0003U)
|
||||
return Function_8000_0003h(Leaf, CPU % PerCPUData.size());
|
||||
else
|
||||
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
bool Hybrid{};
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
|
||||
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
|
||||
using FunctionHandler = std::function<FEXCore::CPUID::FunctionResults()>;
|
||||
void RegisterFunction(uint32_t Function, FunctionHandler Handler) {
|
||||
FunctionHandlers.insert_or_assign(Function, Handler);
|
||||
FunctionHandlers[Function] = Handler;
|
||||
}
|
||||
|
||||
std::unordered_map<uint32_t, FunctionHandler> FunctionHandlers;
|
||||
struct CPUData {
|
||||
const char *ProductName{};
|
||||
#ifdef _M_ARM_64
|
||||
uint32_t MIDR{};
|
||||
#endif
|
||||
bool IsBig{};
|
||||
};
|
||||
std::vector<CPUData> PerCPUData{};
|
||||
|
||||
// Functions
|
||||
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_01h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_02h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_04h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_06h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_07h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf);
|
||||
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf, uint32_t CPU);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf, uint32_t CPU);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf, uint32_t CPU);
|
||||
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0005h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0008h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0009h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
FEXCore::CPUID::FunctionResults Function_0h();
|
||||
FEXCore::CPUID::FunctionResults Function_01h();
|
||||
FEXCore::CPUID::FunctionResults Function_02h();
|
||||
FEXCore::CPUID::FunctionResults Function_06h();
|
||||
FEXCore::CPUID::FunctionResults Function_07h();
|
||||
FEXCore::CPUID::FunctionResults Function_15h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0005h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0006h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0007h();
|
||||
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved();
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,168 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CompileService.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
namespace FEXCore {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::CompileService *This = reinterpret_cast<FEXCore::CompileService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
CompileService::CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, ParentThread {Thread} {
|
||||
|
||||
CompileThreadData = std::make_unique<FEXCore::Core::InternalThreadState>();
|
||||
CompileThreadData->IsCompileService = true;
|
||||
|
||||
// We need a compiler for this work thread
|
||||
CTX->InitializeCompiler(CompileThreadData.get(), true);
|
||||
CompileThreadData->CPUBackend->CopyNecessaryDataForCompileThread(ParentThread->CPUBackend.get());
|
||||
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
}
|
||||
|
||||
void CompileService::Initialize() {
|
||||
// Share CompileService which = this
|
||||
CompileThreadData->CompileService = ParentThread->CompileService;
|
||||
}
|
||||
|
||||
void CompileService::Shutdown() {
|
||||
ShuttingDown = true;
|
||||
// Kick the working thread
|
||||
StartWork.NotifyAll();
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
|
||||
void CompileService::ClearCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// On cache clear we need to spin down the execution thread to ensure it isn't trying to give us more work items
|
||||
if (CompileMutex.try_lock()) {
|
||||
// We can only clear these things if we pulled the compile mutex
|
||||
|
||||
// Grab the work queue and clear it
|
||||
// We don't need to grab the queue mutex since this thread will no longer receive any work events
|
||||
// Threads are bounded 1:1
|
||||
while (WorkQueue.size()) {
|
||||
WorkItem *Item = WorkQueue.front();
|
||||
WorkQueue.pop();
|
||||
delete Item;
|
||||
}
|
||||
|
||||
// Go through the garbage collection array and clear it
|
||||
// It's safe to clear things that aren't marked safe since we are clearing cache
|
||||
if (GCArray.size()) {
|
||||
// Clean up our GC array
|
||||
for (auto it = GCArray.begin(); it != GCArray.end();) {
|
||||
delete *it;
|
||||
it = GCArray.erase(it);
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Throw::A(CompileThreadData->LocalIRCache.size() == 0, "Compile service must never have LocalIRCache");
|
||||
|
||||
CompileMutex.unlock();
|
||||
}
|
||||
|
||||
// Clear the inverse cache of what is calling us from the Context ClearCache routine
|
||||
auto SelectedThread = Thread->IsCompileService ? ParentThread : Thread;
|
||||
SelectedThread->LookupCache->ClearCache();
|
||||
SelectedThread->CPUBackend->ClearCache();
|
||||
}
|
||||
|
||||
|
||||
CompileService::WorkItem *CompileService::CompileCode(uint64_t RIP) {
|
||||
// Tell the worker thread to compile code for us
|
||||
WorkItem *Item = new WorkItem{};
|
||||
Item->RIP = RIP;
|
||||
|
||||
{
|
||||
// Fill the threads work queue
|
||||
std::scoped_lock<std::mutex> lk(QueueMutex);
|
||||
WorkQueue.emplace(Item);
|
||||
}
|
||||
|
||||
// Notify the thread that it has more work
|
||||
StartWork.NotifyAll();
|
||||
|
||||
return Item;
|
||||
}
|
||||
|
||||
void CompileService::ExecutionThread() {
|
||||
// Ignore signals coming from the guest
|
||||
CTX->SignalDelegation->MaskThreadSignals();
|
||||
|
||||
// Set our thread name so we can see its relation
|
||||
char ThreadName[16]{};
|
||||
snprintf(ThreadName, 16, "%ld-CS", ParentThread->ThreadManager.TID.load());
|
||||
pthread_setname_np(pthread_self(), ThreadName);
|
||||
|
||||
while (true) {
|
||||
// Wait for work
|
||||
StartWork.Wait();
|
||||
if (ShuttingDown.load()) {
|
||||
break;
|
||||
}
|
||||
std::scoped_lock<std::mutex> lk(CompileMutex);
|
||||
|
||||
size_t WorkItems{};
|
||||
|
||||
do {
|
||||
// Grab a work item
|
||||
WorkItem *Item{};
|
||||
{
|
||||
std::scoped_lock<std::mutex> lk(QueueMutex);
|
||||
WorkItems = WorkQueue.size();
|
||||
if (WorkItems) {
|
||||
Item = WorkQueue.front();
|
||||
WorkQueue.pop();
|
||||
}
|
||||
}
|
||||
|
||||
// If we had a work item then work on it
|
||||
if (Item) {
|
||||
// Make sure it's not in lookup cache by accident
|
||||
LogMan::Throw::A(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
|
||||
|
||||
// Code isn't in cache, compile now
|
||||
// Set our thread state's RIP
|
||||
CompileThreadData->CurrentFrame->State.rip = Item->RIP;
|
||||
|
||||
auto [CodePtr, IRList, DebugData, RAData, Generated, StartAddr, Length] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
|
||||
|
||||
LogMan::Throw::A(Generated == true, "Compile Service doesn't have IR Cache");
|
||||
|
||||
if (!CodePtr) {
|
||||
// XXX: We currently have the expectation that compile service code will be significantly smaller than regular thread's code
|
||||
ERROR_AND_DIE("Couldn't compile code for thread at RIP: 0x%lx", Item->RIP);
|
||||
}
|
||||
|
||||
Item->CodePtr = CodePtr;
|
||||
Item->IRList = IRList;
|
||||
Item->DebugData = DebugData;
|
||||
Item->RAData = RAData;
|
||||
Item->StartAddr = StartAddr;
|
||||
Item->Length = Length;
|
||||
|
||||
GCArray.emplace_back(Item);
|
||||
Item->ServiceWorkDone.NotifyAll();
|
||||
}
|
||||
} while (WorkItems != 0);
|
||||
|
||||
if (GCArray.size()) {
|
||||
// Clean up our GC array
|
||||
for (auto it = GCArray.begin(); it != GCArray.end();) {
|
||||
if ((*it)->SafeToClear) {
|
||||
delete *it;
|
||||
it = GCArray.erase(it);
|
||||
}
|
||||
else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <memory>
|
||||
#include <thread>
|
||||
#include <unordered_map>
|
||||
#include <queue>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
struct Context;
|
||||
}
|
||||
namespace Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace IR {
|
||||
class RegisterAllocationData;
|
||||
};
|
||||
class CompileService final {
|
||||
public:
|
||||
CompileService(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
void Initialize();
|
||||
void Shutdown();
|
||||
|
||||
struct WorkItem {
|
||||
// Incoming
|
||||
uint64_t RIP{};
|
||||
|
||||
// Outgoing
|
||||
void *CodePtr{};
|
||||
FEXCore::IR::IRListView *IRList{};
|
||||
FEXCore::IR::RegisterAllocationData *RAData{};
|
||||
FEXCore::Core::DebugData *DebugData{};
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
|
||||
// Communication
|
||||
Event ServiceWorkDone{};
|
||||
std::atomic_bool SafeToClear{};
|
||||
};
|
||||
|
||||
WorkItem *CompileCode(uint64_t RIP);
|
||||
void ClearCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ParentThread;
|
||||
|
||||
std::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::unique_ptr<FEXCore::Core::InternalThreadState> CompileThreadData;
|
||||
|
||||
std::mutex QueueMutex{};
|
||||
std::mutex CompileMutex{};
|
||||
std::queue<WorkItem*> WorkQueue{};
|
||||
std::vector<WorkItem*> GCArray{};
|
||||
Event StartWork{};
|
||||
std::atomic_bool ShuttingDown{false};
|
||||
};
|
||||
}
|
||||
+728
-703
File diff suppressed because it is too large.
Load diff
+126
-386
@@ -1,37 +1,15 @@
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <code-buffer-vixl.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
MemOperand(STATE, offsetof(FEXCore::Core::STATE_TYPE, FIELD))
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -41,11 +19,13 @@ using namespace vixl::aarch64;
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread), Arm64Emitter(MAX_DISPATCHER_CODE_SIZE) {
|
||||
SRAEnabled = config.StaticRegisterAssignment;
|
||||
SetAllowAssembler(true);
|
||||
|
||||
DispatchPtr = GetCursorAddress<AsmDispatch>();
|
||||
auto Buffer = GetBuffer();
|
||||
DispatchPtr = Buffer->GetOffsetAddress<CPUBackend::AsmDispatch>(GetCursorOffset());
|
||||
|
||||
// while (true) {
|
||||
// Ptr = FindBlock(RIP)
|
||||
@@ -55,9 +35,15 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
// Ptr();
|
||||
// }
|
||||
|
||||
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
|
||||
Literal l_VirtualMemory {VirtualMemorySize};
|
||||
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
|
||||
Literal l_L1Ptr {Thread->LookupCache->GetL1Pointer()};
|
||||
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
|
||||
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
|
||||
Literal l_CompileBlock {GetCompileBlockPtr()};
|
||||
Literal l_ExitFunctionLink {config.ExitFunctionLink};
|
||||
Literal l_ExitFunctionLinkThis {config.ExitFunctionLinkThis};
|
||||
|
||||
// Push all the register we need to save
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -70,18 +56,17 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
// regardless of where we were in the stack
|
||||
add(x0, sp, 0);
|
||||
str(x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
|
||||
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
|
||||
AbsoluteLoopTopAddressFillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
|
||||
if (config.StaticRegisterAllocation) {
|
||||
if (SRAEnabled) {
|
||||
FillStaticRegs();
|
||||
}
|
||||
|
||||
// We want to ensure that we are 16 byte aligned at the top of this loop
|
||||
Align16B();
|
||||
aarch64::Label FullLookup{};
|
||||
aarch64::Label CallBlock{};
|
||||
aarch64::Label LoopTop{};
|
||||
aarch64::Label ExitSpillSRA{};
|
||||
aarch64::Label ThreadPauseHandler{};
|
||||
@@ -91,34 +76,34 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify x2 since it contains our RIP once the block doesn't exist
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.rip));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
auto RipReg = x2;
|
||||
|
||||
// L1 Cache
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
// L1 Cache
|
||||
ldr(x0, &l_L1Ptr);
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
ldp(x3, x0, MemOperand(x0));
|
||||
cmp(x0, RipReg);
|
||||
b(&FullLookup, Condition::ne);
|
||||
|
||||
br(x3);
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
ldp(x1, x0, MemOperand(x0));
|
||||
cmp(x0, RipReg);
|
||||
b(&FullLookup, Condition::ne);
|
||||
br(x1);
|
||||
}
|
||||
|
||||
// L1C check failed, do a full lookup
|
||||
bind(&FullLookup);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
ldr(x0, &l_PagePtr);
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(x3, RipReg, VirtualMemorySize - 1);
|
||||
if (__builtin_popcountl(VirtualMemorySize) == 1) {
|
||||
and_(x3, RipReg, Thread->LookupCache->GetVirtualMemorySize() - 1);
|
||||
}
|
||||
else {
|
||||
LoadConstant(x3, VirtualMemorySize);
|
||||
ldr(x3, &l_VirtualMemory);
|
||||
and_(x3, RipReg, x3);
|
||||
}
|
||||
|
||||
@@ -151,25 +136,51 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x1, Shift::LSL, 4));
|
||||
stp(x3, x2, MemOperand(x0));
|
||||
|
||||
// Jump to the block
|
||||
br(x3);
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
// update L1 cache
|
||||
ldr(x0, &l_L1Ptr);
|
||||
|
||||
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x1, Shift::LSL, 4));
|
||||
stp(x3, x2, MemOperand(x0));
|
||||
|
||||
br(x3);
|
||||
} else {
|
||||
mov(x0, STATE);
|
||||
blr(x3);
|
||||
}
|
||||
}
|
||||
|
||||
if (config.ExecuteBlocksWithCall) {
|
||||
// Interpreter continues execution here
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
ldr(x0, &l_CTX);
|
||||
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
// If the value == 0 then branch to the top
|
||||
cbz(x0, &LoopTop);
|
||||
// Else we need to pause now
|
||||
b(&ThreadPauseHandler);
|
||||
}
|
||||
else {
|
||||
// Unconditionally loop to the top
|
||||
// We will only stop on error when compiling a block or signal
|
||||
b(&LoopTop);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
bind(&ExitSpillSRA);
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
ThreadStopHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
ThreadStopHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
|
||||
PopCalleeSavedRegisters();
|
||||
|
||||
@@ -178,56 +189,19 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
ExitFunctionLinkerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
ldr(x0, &l_ExitFunctionLinkThis);
|
||||
mov(x1, STATE);
|
||||
mov(x2, lr);
|
||||
|
||||
LoadConstant(x0, ~0ULL);
|
||||
stp(x0, x0, MemOperand(sp, -16, PreIndex));
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
add(x2, sp, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
}
|
||||
|
||||
mov(x0, STATE);
|
||||
mov(x1, lr);
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
ldr(x3, &l_ExitFunctionLink);
|
||||
blr(x3);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
|
||||
mov(x4, x0);
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
LoadConstant(x2, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(sp, sp, 16);
|
||||
|
||||
mov(x0, x4);
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
if (SRAEnabled)
|
||||
FillStaticRegs();
|
||||
br(x0);
|
||||
}
|
||||
@@ -236,61 +210,24 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
{
|
||||
bind(&NoBlock);
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(x0, ~0ULL);
|
||||
stp(x0, x2, MemOperand(sp, -16, PreIndex));
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
add(x2, sp, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Reload x2 to bring back RIP
|
||||
ldr(x2, MemOperand(sp, 8, Offset));
|
||||
}
|
||||
|
||||
ldr(x0, &l_CTX);
|
||||
mov(x1, STATE);
|
||||
ldr(x3, &l_CompileBlock);
|
||||
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
// X2 contains our guest RIP
|
||||
blr(x3); // { CTX, Frame, RIP}
|
||||
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
LoadConstant(x0, SIG_SETMASK);
|
||||
add(x1, sp, 0);
|
||||
LoadConstant(x2, 0);
|
||||
LoadConstant(x3, 8);
|
||||
LoadConstant(x8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(sp, sp, 16);
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
if (SRAEnabled)
|
||||
FillStaticRegs();
|
||||
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
SignalHandlerReturnAddress = GetCursorAddress<uint64_t>();
|
||||
SignalHandlerReturnAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
|
||||
// Now to get back to our old location we need to do a fault dance
|
||||
// We can't use SIGTRAP here since gdb catches it and never gives it to the application!
|
||||
@@ -298,50 +235,12 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
}
|
||||
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest SIGTRAP handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
brk(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
LoadConstant(x1, 0);
|
||||
ldr(x1, MemOperand(x1));
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
ThreadPauseHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
bind(&ThreadPauseHandler);
|
||||
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
ThreadPauseHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
// We are pausing, this means the frontend should be waiting for this thread to idle
|
||||
// We will have faulted and jumped to this location at this point
|
||||
|
||||
@@ -351,7 +250,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ldr(x2, &l_Sleep);
|
||||
blr(x2);
|
||||
|
||||
PauseReturnInstruction = GetCursorAddress<uint64_t>();
|
||||
PauseReturnInstruction = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
// Fault to start running again
|
||||
hlt(0);
|
||||
}
|
||||
@@ -372,7 +271,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
// On return to the thunk, the thunk can get whatever its return value is from the thread context depending on ABI handling on its end
|
||||
// When the thunk itself returns, it'll do its regular return logic there
|
||||
// void ReentrantCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
CallbackPtr = GetCursorAddress<JITCallback>();
|
||||
CallbackPtr = Buffer->GetOffsetAddress<CPUBackend::JITCallback>(GetCursorOffset());
|
||||
|
||||
// We expect the thunk to have previously pushed the registers it was using
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -381,243 +280,84 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
mov(STATE, x0);
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
ldr(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
|
||||
ldr(w2, MemOperand(x0));
|
||||
add(w2, w2, 1);
|
||||
str(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
str(w2, MemOperand(x0));
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
LoadConstant(x0, CTX->X86CodeGen.CallbackReturn);
|
||||
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
sub(x2, x2, 16);
|
||||
str(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
str(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
// Guest stack is now correctly misaligned after a regular call instruction
|
||||
str(x0, MemOperand(x2));
|
||||
|
||||
// Store RIP to the context state
|
||||
str(x1, STATE_PTR(CpuStateFrame, State.rip));
|
||||
str(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
|
||||
// load static regs
|
||||
if (config.StaticRegisterAllocation)
|
||||
if (SRAEnabled)
|
||||
FillStaticRegs();
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
{
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
}
|
||||
|
||||
place(&l_VirtualMemory);
|
||||
place(&l_PagePtr);
|
||||
place(&l_L1Ptr);
|
||||
place(&l_CTX);
|
||||
place(&l_Sleep);
|
||||
place(&l_CompileBlock);
|
||||
place(&l_ExitFunctionLink);
|
||||
place(&l_ExitFunctionLinkThis);
|
||||
|
||||
|
||||
FinalizeCode();
|
||||
Start = reinterpret_cast<uint64_t>(DispatchPtr);
|
||||
End = GetCursorAddress<uint64_t>();
|
||||
End = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
GetBuffer()->SetExecutable();
|
||||
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
|
||||
#if ENABLE_JITSYMBOLS
|
||||
std::string Name = "Dispatch_" + std::to_string(::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(void *ucontext) {
|
||||
for(int i = 0; i < SRA64.size(); i++) {
|
||||
ThreadState->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
// TODO: Also recover FPRs, not sure where the neon context is
|
||||
// This is usually not needed
|
||||
/*
|
||||
for(int i = 0; i < SRAFPR.size(); i++) {
|
||||
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
|
||||
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
|
||||
}
|
||||
*/
|
||||
}
|
||||
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline, destination buffer is set before use
|
||||
static thread_local vixl::aarch64::Assembler emit((uint8_t*)&emit, 1);
|
||||
#ifdef _M_ARM_64
|
||||
|
||||
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
DispatcherConfig config;
|
||||
config.ExecuteBlocksWithCall = true;
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxGDBPauseCheckSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Dispatcher = new Arm64Dispatcher(ctx, Thread, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
aarch64::Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(FEXCore::Context::Context::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Thread)); // Get thread
|
||||
emit.ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
|
||||
emit.ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cbz(w0, &RunBlock);
|
||||
{
|
||||
Literal l_GuestRIP {GuestRIP};
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.ldr(x0, &l_GuestRIP);
|
||||
emit.str(x0, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Stop the thread
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.br(x0);
|
||||
emit.place(&l_GuestRIP);
|
||||
}
|
||||
emit.bind(&RunBlock);
|
||||
emit.FinalizeCode();
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
|
||||
}
|
||||
|
||||
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxInterpreterTrampolineSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label InlineIRData;
|
||||
|
||||
emit.mov(x0, STATE);
|
||||
emit.adr(x1, &InlineIRData);
|
||||
|
||||
emit.ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
emit.blr(x3);
|
||||
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
emit.br(x0);
|
||||
|
||||
emit.bind(&InlineIRData);
|
||||
|
||||
emit.FinalizeCode();
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
|
||||
for (size_t i = 0; i < SRA64.size(); i++) {
|
||||
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
|
||||
// Skip this one, it's already spilled
|
||||
continue;
|
||||
}
|
||||
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
|
||||
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
|
||||
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
|
||||
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
|
||||
AArch64.LUDIVHandler = LUDIVHandlerAddress;
|
||||
AArch64.LDIVHandler = LDIVHandlerAddress;
|
||||
AArch64.LUREMHandler = LUREMHandlerAddress;
|
||||
AArch64.LREMHandler = LREMHandlerAddress;
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<Arm64Dispatcher>(CTX, Config);
|
||||
}
|
||||
#endif
|
||||
|
||||
}
|
||||
@@ -3,32 +3,16 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
public:
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
|
||||
|
||||
protected:
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
|
||||
|
||||
private:
|
||||
// Long division helpers
|
||||
uint64_t LUDIVHandlerAddress{};
|
||||
uint64_t LDIVHandlerAddress{};
|
||||
uint64_t LUREMHandlerAddress{};
|
||||
uint64_t LREMHandlerAddress{};
|
||||
void SpillSRA(void *ucontext) override;
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
+111
-601
@@ -1,24 +1,8 @@
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include "Common/MathUtils.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <csignal>
|
||||
#include <cstring>
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -28,19 +12,15 @@ void Dispatcher::SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuS
|
||||
--ctx->IdleWaitRefCount;
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
|
||||
Thread->RunningEvents.ThreadSleeping = true;
|
||||
|
||||
// Go to sleep
|
||||
Thread->StartRunning.Wait();
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
++ctx->IdleWaitRefCount;
|
||||
Thread->RunningEvents.ThreadSleeping = false;
|
||||
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
}
|
||||
|
||||
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext) {
|
||||
void Dispatcher::StoreThreadState(int Signal, void *ucontext) {
|
||||
// We can end up getting a signal at any point in our host state
|
||||
// Jump to a handler that saves all state so we can safely return
|
||||
uint64_t OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
@@ -65,414 +45,89 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(FEXCore::Core:
|
||||
// Save guest state
|
||||
// We can't guarantee if registers are in context or host GPRs
|
||||
// So we need to save everything
|
||||
memcpy(&Context->GuestState, Thread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(&Context->GuestState, ThreadState->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Set the new SP
|
||||
ArchHelpers::Context::SetSp(ucontext, NewSP);
|
||||
|
||||
// Signal frames are only used on the interpreter
|
||||
// The JITS require the stack to be setup correctly on rt_sigreturn
|
||||
if (CTX->Config.Core() == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
SignalFrames.push(NewSP);
|
||||
}
|
||||
|
||||
Context->Flags = 0;
|
||||
Context->FPStateLocation = 0;
|
||||
Context->UContextLocation = 0;
|
||||
Context->SigInfoLocation = 0;
|
||||
|
||||
// Store fault to top status and then reset it
|
||||
Context->FaultToTopAndGeneratedException = Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
|
||||
return Context;
|
||||
SignalFrames.push(NewSP);
|
||||
}
|
||||
|
||||
void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext) {
|
||||
uint64_t OldSP{};
|
||||
if (CTX->Config.Core() == FEXCore::Config::CONFIG_IRJIT) {
|
||||
OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(!SignalFrames.empty(), "Trying to restore a signal frame when we don't have any");
|
||||
OldSP = SignalFrames.top();
|
||||
SignalFrames.pop();
|
||||
}
|
||||
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
uint64_t OldSP = SignalFrames.top();
|
||||
SignalFrames.pop();
|
||||
uintptr_t NewSP = OldSP;
|
||||
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
|
||||
|
||||
// First thing, reset the guest state
|
||||
memcpy(Thread->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(ThreadState->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Now restore host state
|
||||
ArchHelpers::Context::RestoreContext(ucontext, Context);
|
||||
|
||||
if (Context->UContextLocation) {
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (Context->Flags &ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT) {
|
||||
// XXX: Unsupported since it needs state reconstruction
|
||||
// If we are in the JIT then SRA might need to be restored to values from the context
|
||||
// We can't currently support this since it might result in tearing without real state reconstruction
|
||||
}
|
||||
|
||||
if (!(Context->Flags & ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT)) {
|
||||
auto *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(Context->UContextLocation);
|
||||
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<siginfo_t*>(Context->SigInfoLocation);
|
||||
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] ||
|
||||
Context->FaultToTopAndGeneratedException) {
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
|
||||
// XXX: Full context setting
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
Frame->State.flags[1] = 1;
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x];
|
||||
COPY_REG(R8);
|
||||
COPY_REG(R9);
|
||||
COPY_REG(R10);
|
||||
COPY_REG(R11);
|
||||
COPY_REG(R12);
|
||||
COPY_REG(R13);
|
||||
COPY_REG(R14);
|
||||
COPY_REG(R15);
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][0], &fpstate->_xmm[i], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][2], &xstate->ymmh.ymmh_space[i], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.FTW = fpstate->ftw;
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(Context->UContextLocation);
|
||||
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(Context->SigInfoLocation);
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] ||
|
||||
Context->FaultToTopAndGeneratedException) {
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
// XXX: Full context setting
|
||||
// First 32-bytes of flags is EFLAGS broken out
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
Frame->State.flags[1] = 1;
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
|
||||
Frame->State.cs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.FTW = fpstate->ftw;
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Restore the previous signal state
|
||||
// This allows recursive signals to properly handle signal masking as we are walking back up the list of signals
|
||||
CTX->SignalDelegation->SetCurrentSignal(Context->Signal);
|
||||
}
|
||||
|
||||
static uint32_t ConvertSignalToTrapNo(int Signal, siginfo_t *HostSigInfo) {
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
if (HostSigInfo->si_code == SEGV_MAPERR ||
|
||||
HostSigInfo->si_code == SEGV_ACCERR) {
|
||||
// Protection fault
|
||||
return X86State::X86_TRAPNO_PF;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// Unknown mapping, fall back to old behaviour and just pass signal
|
||||
return Signal;
|
||||
}
|
||||
|
||||
static uint32_t ConvertSignalToError(int Signal, siginfo_t *HostSigInfo) {
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
if (HostSigInfo->si_code == SEGV_MAPERR ||
|
||||
HostSigInfo->si_code == SEGV_ACCERR) {
|
||||
// Protection fault
|
||||
// Always a user fault for us
|
||||
// XXX: PF_PROT and PF_WRITE
|
||||
return X86State::X86_PF_USER;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// Not a page fault issue
|
||||
return 0;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void SetXStateInfo(T* xstate, bool is_avx_enabled) {
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
fpstate->sw_reserved.magic1 = x86_64::fpx_sw_bytes::FP_XSTATE_MAGIC;
|
||||
fpstate->sw_reserved.extended_size = is_avx_enabled ? sizeof(T) : 0;
|
||||
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_FP |
|
||||
x86_64::fpx_sw_bytes::FEATURE_SSE;
|
||||
if (is_avx_enabled) {
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_YMM;
|
||||
}
|
||||
|
||||
fpstate->sw_reserved.xstate_size = fpstate->sw_reserved.extended_size;
|
||||
|
||||
if (is_avx_enabled) {
|
||||
xstate->xstate_hdr.xfeatures = 0;
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
auto ContextBackup = StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
StoreThreadState(Signal, ucontext);
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
++SignalHandlerRefCounter;
|
||||
|
||||
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
|
||||
// Set the new PC
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
|
||||
uint64_t OldGuestSP = Frame->State.gregs[X86State::REG_RSP];
|
||||
uint64_t NewGuestSP = OldGuestSP;
|
||||
|
||||
// Pulling from context here
|
||||
const bool Is64BitMode = CTX->Config.Is64BitMode;
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
const uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
|
||||
|
||||
// Spill the SRA regardless of signal handler type
|
||||
// We are going to be returning to the top of the dispatcher which will fill again
|
||||
// Otherwise we might load garbage
|
||||
if (config.StaticRegisterAllocation) {
|
||||
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
|
||||
uint32_t IgnoreMask{};
|
||||
#ifdef _M_ARM_64
|
||||
if (Frame->InSyscallInfo != 0) {
|
||||
// We are in a syscall, this means we are in a weird register state
|
||||
// We need to spill SRA but only some of it, since some values have already been spilled
|
||||
// Lower 16 bits tells us which registers are already spilled to the context
|
||||
// So we ignore spilling those ones
|
||||
uint16_t NumRegisters = std::popcount(Frame->InSyscallInfo & 0xFFFF);
|
||||
if (NumRegisters >= 4) {
|
||||
// Unhandled case
|
||||
IgnoreMask = 0;
|
||||
}
|
||||
else {
|
||||
IgnoreMask = Frame->InSyscallInfo & 0xFFFF;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We must spill everything
|
||||
IgnoreMask = 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
// We are in jit, SRA must be spilled
|
||||
SpillSRA(Thread, ucontext, IgnoreMask);
|
||||
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT;
|
||||
} else {
|
||||
if (!IsAddressInDispatcher(OldPC)) {
|
||||
// This is likely to cause issues but in some cases it isn't fatal
|
||||
// This can also happen if we have put a signal on hold, then we just reenabled the signal
|
||||
// So we are in the syscall handler
|
||||
// Only throw a log message in this case
|
||||
if constexpr (false) {
|
||||
// XXX: Messages in the signal handler can cause us to crash
|
||||
LogMan::Msg::EFmt("Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
}
|
||||
if (!(GuestStack->ss_flags & SS_DISABLE)) {
|
||||
// If our guest is already inside of the alternative stack
|
||||
// Then that means we are hitting recursive signals and we need to walk back the stack correctly
|
||||
uint64_t AltStackBase = reinterpret_cast<uint64_t>(GuestStack->ss_sp);
|
||||
uint64_t AltStackEnd = AltStackBase + GuestStack->ss_size;
|
||||
if (OldGuestSP >= AltStackBase &&
|
||||
OldGuestSP <= AltStackEnd) {
|
||||
// We are already in the alt stack, the rest of the code will handle adjusting this
|
||||
}
|
||||
else {
|
||||
NewGuestSP = AltStackEnd;
|
||||
}
|
||||
}
|
||||
|
||||
// altstack is only used if the signal handler was setup with SA_ONSTACK
|
||||
if (GuestAction->sa_flags & SA_ONSTACK) {
|
||||
// Additionally the altstack is only used if the enabled (SS_DISABLE flag is not set)
|
||||
if (!(GuestStack->ss_flags & SS_DISABLE)) {
|
||||
// If our guest is already inside of the alternative stack
|
||||
// Then that means we are hitting recursive signals and we need to walk back the stack correctly
|
||||
uint64_t AltStackBase = reinterpret_cast<uint64_t>(GuestStack->ss_sp);
|
||||
uint64_t AltStackEnd = AltStackBase + GuestStack->ss_size;
|
||||
if (OldGuestSP >= AltStackBase &&
|
||||
OldGuestSP <= AltStackEnd) {
|
||||
// We are already in the alt stack, the rest of the code will handle adjusting this
|
||||
}
|
||||
else {
|
||||
NewGuestSP = AltStackEnd;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Is64BitMode) {
|
||||
// Back up past the redzone, which is 128bytes
|
||||
// 32-bit doesn't have a redzone
|
||||
NewGuestSP -= 128;
|
||||
}
|
||||
|
||||
// siginfo_t
|
||||
siginfo_t *HostSigInfo = reinterpret_cast<siginfo_t*>(info);
|
||||
|
||||
// Backup where we think the RIP currently is
|
||||
ContextBackup->OriginalRIP = Frame->State.rip;
|
||||
// Back up past the redzone, which is 128bytes
|
||||
// Don't need this offset if we aren't going to be putting siginfo in to it
|
||||
NewGuestSP -= 128;
|
||||
|
||||
if (GuestAction->sa_flags & SA_SIGINFO) {
|
||||
// Setup ucontext a bit
|
||||
if (Is64BitMode) {
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86_64::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::xstate));
|
||||
if (SRAEnabled) {
|
||||
if (!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
LogMan::Throw::A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86_64::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::_libc_fpstate));
|
||||
// We are in jit, SRA must be spilled
|
||||
SpillSRA(ucontext);
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
}
|
||||
|
||||
// Setup ucontext a bit
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::ucontext_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::ucontext_t));
|
||||
uint64_t UContextLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(siginfo_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(siginfo_t));
|
||||
uint64_t SigInfoLocation = NewGuestSP;
|
||||
|
||||
ContextBackup->FPStateLocation = FPStateLocation;
|
||||
ContextBackup->UContextLocation = UContextLocation;
|
||||
ContextBackup->SigInfoLocation = SigInfoLocation;
|
||||
|
||||
FEXCore::x86_64::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(UContextLocation);
|
||||
siginfo_t *guest_siginfo = reinterpret_cast<siginfo_t*>(SigInfoLocation);
|
||||
|
||||
// We have extended float information
|
||||
guest_uctx->uc_flags = FEXCore::x86_64::UC_FP_XSTATE;
|
||||
guest_uctx->uc_flags |= FEXCore::x86_64::UC_FP_XSTATE;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CSGSFS] = 0;
|
||||
|
||||
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
|
||||
// SI_USER could also potentially have random data in it, needs to be bit perfect
|
||||
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
|
||||
// Overwrite si_code
|
||||
guest_siginfo->si_code = Thread->CurrentFrame->SynchronousFaultData.si_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_OLDMASK] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CR2] = 0;
|
||||
guest_uctx->uc_mcontext.fpregs = &guest_uctx->__fpregs_mem;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
|
||||
@@ -494,28 +149,15 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
memcpy(guest_uctx->__fpregs_mem._st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
memcpy(guest_uctx->__fpregs_mem._xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
fpstate->ftw = Frame->State.FTW;
|
||||
guest_uctx->__fpregs_mem.fcw = Frame->State.FCW;
|
||||
|
||||
// Reconstruct FSW
|
||||
fpstate->fsw =
|
||||
guest_uctx->__fpregs_mem.fsw =
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
|
||||
@@ -527,291 +169,136 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
guest_uctx->uc_stack.ss_sp = GuestStack->ss_sp;
|
||||
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
|
||||
|
||||
Frame->State.gregs[X86State::REG_RSI] = SigInfoLocation;
|
||||
// XXX: siginfo_t(RSI)
|
||||
Frame->State.gregs[X86State::REG_RSI] = 0x4142434445460000;
|
||||
Frame->State.gregs[X86State::REG_RDX] = UContextLocation;
|
||||
}
|
||||
else {
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT;
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::_libc_fpstate));
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
// XXX: 32bit Support
|
||||
NewGuestSP -= sizeof(FEXCore::x86::ucontext_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::ucontext_t));
|
||||
uint64_t UContextLocation = NewGuestSP;
|
||||
|
||||
uint64_t UContextLocation = 0; // NewGuestSP;
|
||||
NewGuestSP -= sizeof(FEXCore::x86::siginfo_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::siginfo_t));
|
||||
uint64_t SigInfoLocation = NewGuestSP;
|
||||
|
||||
ContextBackup->FPStateLocation = FPStateLocation;
|
||||
ContextBackup->UContextLocation = UContextLocation;
|
||||
ContextBackup->SigInfoLocation = SigInfoLocation;
|
||||
|
||||
FEXCore::x86::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(UContextLocation);
|
||||
FEXCore::x86::siginfo_t *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(SigInfoLocation);
|
||||
|
||||
// We have extended float information
|
||||
guest_uctx->uc_flags = FEXCore::x86::UC_FP_XSTATE;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = static_cast<uint32_t>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
fpstate->ftw = Frame->State.FTW;
|
||||
// Reconstruct FSW
|
||||
fpstate->fsw =
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] << 10) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] << 14);
|
||||
|
||||
// Copy over signal stack information
|
||||
guest_uctx->uc_stack.ss_flags = GuestStack->ss_flags;
|
||||
guest_uctx->uc_stack.ss_sp = static_cast<uint32_t>(reinterpret_cast<uint64_t>(GuestStack->ss_sp));
|
||||
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
|
||||
|
||||
// These three elements are in every siginfo
|
||||
guest_siginfo->si_signo = HostSigInfo->si_signo;
|
||||
guest_siginfo->si_errno = HostSigInfo->si_errno;
|
||||
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
case SIGBUS:
|
||||
// Macro expansion to get the si_addr
|
||||
// This is the address trying to be accessed, not the RIP
|
||||
guest_siginfo->_sifields._sigfault.addr = static_cast<uint32_t>(reinterpret_cast<uintptr_t>(HostSigInfo->si_addr));
|
||||
break;
|
||||
case SIGFPE:
|
||||
case SIGILL:
|
||||
// Macro expansion to get the si_addr
|
||||
// Can't really give a real result here. Pull from the context for now
|
||||
guest_siginfo->_sifields._sigfault.addr = Frame->State.rip;
|
||||
break;
|
||||
case SIGCHLD:
|
||||
guest_siginfo->_sifields._sigchld.pid = HostSigInfo->si_pid;
|
||||
guest_siginfo->_sifields._sigchld.uid = HostSigInfo->si_uid;
|
||||
guest_siginfo->_sifields._sigchld.status = HostSigInfo->si_status;
|
||||
guest_siginfo->_sifields._sigchld.utime = HostSigInfo->si_utime;
|
||||
guest_siginfo->_sifields._sigchld.stime = HostSigInfo->si_stime;
|
||||
break;
|
||||
case SIGALRM:
|
||||
case SIGVTALRM:
|
||||
guest_siginfo->_sifields._timer.tid = HostSigInfo->si_timerid;
|
||||
guest_siginfo->_sifields._timer.overrun = HostSigInfo->si_overrun;
|
||||
guest_siginfo->_sifields._timer.sigval.sival_int = HostSigInfo->si_int;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled siginfo_t for signal: {}\n", Signal);
|
||||
break;
|
||||
}
|
||||
uint64_t SigInfoLocation = 0; // NewGuestSP;
|
||||
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = UContextLocation;
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = SigInfoLocation;
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = Signal;
|
||||
}
|
||||
|
||||
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.sigaction);
|
||||
}
|
||||
else {
|
||||
if (!Is64BitMode) {
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = Signal;
|
||||
}
|
||||
|
||||
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.handler);
|
||||
}
|
||||
|
||||
if (Is64BitMode) {
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RDI] = Signal;
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
Frame->State.gregs[X86State::REG_RDI] = Signal;
|
||||
|
||||
// Set up the new SP for stack handling
|
||||
NewGuestSP -= 8;
|
||||
*(uint64_t*)NewGuestSP = SignalReturn;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
*(uint64_t*)NewGuestSP = CTX->X86CodeGen.SignalReturn;
|
||||
Frame->State.gregs[X86State::REG_RSP] = NewGuestSP;
|
||||
}
|
||||
else {
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = SignalReturn;
|
||||
LOGMAN_THROW_AA_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
*(uint32_t*)NewGuestSP = CTX->X86CodeGen.SignalReturn;
|
||||
LogMan::Throw::A(CTX->X86CodeGen.SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
Frame->State.gregs[X86State::REG_RSP] = NewGuestSP;
|
||||
}
|
||||
|
||||
// The guest starts its signal frame with a zero initialized FPU
|
||||
// Set that up now. Little bit costly but it's a requirement
|
||||
// This state will be restored on rt_sigreturn
|
||||
memset(Frame->State.xmm.avx.data, 0, sizeof(Frame->State.xmm));
|
||||
memset(Frame->State.mm, 0, sizeof(Frame->State.mm));
|
||||
Frame->State.FCW = 0x37F;
|
||||
Frame->State.FTW = 0xFFFF;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
bool Dispatcher::HandleSIGILL(int Signal, void *info, void *ucontext) {
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == SignalHandlerReturnAddress) {
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
RestoreThreadState(ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
--SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == PauseReturnInstruction) {
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
RestoreThreadState(ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
--SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = Thread->SignalReason.load();
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = ThreadState->SignalReason.load();
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_PAUSE) {
|
||||
// Store our thread state so we can come back to this
|
||||
StoreThreadState(Thread, Signal, ucontext);
|
||||
StoreThreadState(Signal, ucontext);
|
||||
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
if (SRAEnabled) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
LogMan::Throw::A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
|
||||
}
|
||||
|
||||
// Set the new PC
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
|
||||
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
++SignalHandlerRefCounter;
|
||||
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Stop) {
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_STOP) {
|
||||
// Our thread is stopping
|
||||
// We don't care about anything at this point
|
||||
// Set the stack to our starting location when we entered the core and get out safely
|
||||
ArchHelpers::Context::SetSp(ucontext, Frame->ReturningStackLocation);
|
||||
|
||||
// Our ref counting doesn't matter anymore
|
||||
Thread->CurrentFrame->SignalHandlerRefCounter = 0;
|
||||
SignalHandlerRefCounter = 0;
|
||||
|
||||
// Set the new PC
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
if (SRAEnabled) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
LogMan::Throw::A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
|
||||
}
|
||||
|
||||
// We need to be a little bit careful here
|
||||
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
|
||||
// Then we need to ensure we don't double decrement our idle thread counter
|
||||
if (Thread->RunningEvents.ThreadSleeping) {
|
||||
// If the thread was sleeping then its idle counter was decremented
|
||||
// Reincrement it here to not break logic
|
||||
++Thread->CTX->IdleWaitRefCount;
|
||||
}
|
||||
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Return) {
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_RETURN) {
|
||||
RestoreThreadState(ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
--SignalHandlerRefCounter;
|
||||
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -819,15 +306,38 @@ bool Dispatcher::HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, i
|
||||
}
|
||||
|
||||
uint64_t Dispatcher::GetCompileBlockPtr() {
|
||||
using ClassPtrType = void (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
|
||||
using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
|
||||
union PtrCast {
|
||||
ClassPtrType ClassPtr;
|
||||
uintptr_t Data;
|
||||
};
|
||||
|
||||
PtrCast CompileBlockPtr;
|
||||
CompileBlockPtr.ClassPtr = &FEXCore::Context::Context::CompileBlockJit;
|
||||
CompileBlockPtr.ClassPtr = &FEXCore::Context::Context::CompileBlock;
|
||||
return CompileBlockPtr.Data;
|
||||
}
|
||||
|
||||
void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
|
||||
for (auto iter = CodeBuffers.begin(); iter != CodeBuffers.end(); ++iter) {
|
||||
auto [start, end] = *iter;
|
||||
if (start == reinterpret_cast<uint64_t>(start_to_remove)) {
|
||||
CodeBuffers.erase(iter);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher) {
|
||||
for (auto [start, end] : CodeBuffers) {
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
if (IncludeDispatcher) {
|
||||
return IsAddressInDispatcher(Address);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
+36
-65
@@ -1,38 +1,27 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <stack>
|
||||
#include <tuple>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
struct GuestSigAction;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct CpuStateFrame;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
struct DispatcherConfig {
|
||||
bool StaticRegisterAllocation = false;
|
||||
bool ExecuteBlocksWithCall = false;
|
||||
uintptr_t ExitFunctionLink = 0;
|
||||
uintptr_t ExitFunctionLinkThis = 0;
|
||||
bool StaticRegisterAssignment = false;
|
||||
};
|
||||
|
||||
class Dispatcher {
|
||||
public:
|
||||
virtual ~Dispatcher() = default;
|
||||
|
||||
CPUBackend::AsmDispatch DispatchPtr;
|
||||
CPUBackend::JITCallback CallbackPtr;
|
||||
FEXCore::Context::Context::IntCallbackReturn ReturnPtr;
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
@@ -44,70 +33,52 @@ public:
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA{};
|
||||
uint64_t ExitFunctionLinkerAddress{};
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t GuestSignal_SIGILL{};
|
||||
uint64_t GuestSignal_SIGTRAP{};
|
||||
uint64_t GuestSignal_SIGSEGV{};
|
||||
uint64_t IntCallbackReturnAddress{};
|
||||
|
||||
uint64_t PauseReturnInstruction{};
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SignalHandlerRefCounter{};
|
||||
|
||||
uint64_t Start{};
|
||||
uint64_t End{};
|
||||
|
||||
bool HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
bool HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(int Signal, void *info, void *ucontext);
|
||||
|
||||
bool IsAddressInDispatcher(uint64_t Address) const {
|
||||
void RegisterCodeBuffer(uint8_t* start, size_t size) {
|
||||
CodeBuffers.emplace_back(reinterpret_cast<uint64_t>(start),
|
||||
reinterpret_cast<uint64_t>(start + size));
|
||||
}
|
||||
|
||||
void RemoveCodeBuffer(uint8_t* start);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true);
|
||||
bool IsAddressInDispatcher(uint64_t Address) {
|
||||
return Address >= Start && Address < End;
|
||||
}
|
||||
|
||||
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
// These are across all arches for now
|
||||
static constexpr size_t MaxGDBPauseCheckSize = 128;
|
||||
static constexpr size_t MaxInterpreterTrampolineSize = 128;
|
||||
|
||||
virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0;
|
||||
virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0;
|
||||
|
||||
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
DispatchPtr(Frame);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &Config)
|
||||
Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, config {Config}
|
||||
{}
|
||||
, ThreadState {Thread} {}
|
||||
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext);
|
||||
void RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext);
|
||||
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
|
||||
void StoreThreadState(int Signal, void *ucontext);
|
||||
void RestoreThreadState(void *ucontext);
|
||||
std::stack<uint64_t> SignalFrames;
|
||||
|
||||
virtual void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {}
|
||||
bool SRAEnabled = false;
|
||||
virtual void SpillSRA(void *ucontext) {}
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
DispatcherConfig config;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
|
||||
static void SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuStateFrame *Frame);
|
||||
|
||||
static uint64_t GetCompileBlockPtr();
|
||||
|
||||
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
|
||||
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
|
||||
AsmDispatch DispatchPtr;
|
||||
JITCallback CallbackPtr;
|
||||
private:
|
||||
std::vector<std::tuple<uint64_t, uint64_t>> CodeBuffers; // Start, End
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
+90
-264
@@ -1,43 +1,22 @@
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/mman.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
[STATE + offsetof(FEXCore::Core::STATE_TYPE, FIELD)]
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE r14
|
||||
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: Dispatcher(ctx, config)
|
||||
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
|
||||
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
|
||||
nullptr) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "X86 dispatcher does not support SRA");
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread)
|
||||
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE) {
|
||||
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
DispatchPtr = getCurr<AsmDispatch>();
|
||||
DispatchPtr = getCurr<CPUBackend::AsmDispatch>();
|
||||
|
||||
// Temp registers
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
@@ -83,7 +62,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
|
||||
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
// regardless of where we were in the stack
|
||||
mov(qword STATE_PTR(CpuStateFrame, ReturningStackLocation), rsp);
|
||||
mov(qword [rdi + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)], rsp);
|
||||
|
||||
Label LoopTop;
|
||||
Label FullLookup;
|
||||
@@ -96,26 +75,27 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
|
||||
|
||||
{
|
||||
// Load our RIP
|
||||
mov(rdx, qword STATE_PTR(CPUState, rip));
|
||||
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
|
||||
|
||||
// L1 Cache
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
mov(rax, rdx);
|
||||
if (!config.ExecuteBlocksWithCall)
|
||||
{
|
||||
// L1 Cache
|
||||
mov(r13, Thread->LookupCache->GetL1Pointer());
|
||||
mov(rax, rdx);
|
||||
|
||||
and_(rax, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rax, 4);
|
||||
cmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)], rdx);
|
||||
jne(FullLookup);
|
||||
|
||||
jmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)]);
|
||||
and_(rax, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rax, 4);
|
||||
cmp(qword[r13 + rax + 8], rdx);
|
||||
jne(FullLookup);
|
||||
jmp(qword[r13 + rax + 0]);
|
||||
}
|
||||
|
||||
L(FullLookup);
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
mov(r13, Thread->LookupCache->GetPagePointer());
|
||||
|
||||
// Full lookup
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
mov(rax, rdx);
|
||||
mov(rbx, VirtualMemorySize - 1);
|
||||
mov(rbx, Thread->LookupCache->GetVirtualMemorySize() - 1);
|
||||
and_(rax, rbx);
|
||||
shr(rax, 12);
|
||||
|
||||
@@ -142,15 +122,39 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
|
||||
je(NoBlock);
|
||||
|
||||
// Update L1
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
mov(rcx, rdx);
|
||||
and_(rcx, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rcx, 1);
|
||||
mov(qword[r13 + rcx*8 + 8], rdx);
|
||||
mov(qword[r13 + rcx*8 + 0], rax);
|
||||
if (config.ExecuteBlocksWithCall) {
|
||||
mov(r13, Thread->LookupCache->GetL1Pointer());
|
||||
mov(rcx, rdx);
|
||||
and_(rcx, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rcx, 1);
|
||||
mov(qword[r13 + rcx*8 + 8], rdx);
|
||||
mov(qword[r13 + rcx*8 + 0], rax);
|
||||
}
|
||||
|
||||
// Real block if we made it here
|
||||
jmp(rax);
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
jmp(rax);
|
||||
} else {
|
||||
mov(rdi, STATE);
|
||||
call(rax);
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
mov(rax, qword [STATE + (offsetof(FEXCore::Core::InternalThreadState, CTX))]);
|
||||
|
||||
// If the value == 0 then branch to the top
|
||||
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
je(LoopTop);
|
||||
// Else we need to pause now
|
||||
jmp(ThreadPauseHandler);
|
||||
ud2();
|
||||
}
|
||||
else {
|
||||
jmp(LoopTop);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
@@ -169,122 +173,40 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
// Block creation
|
||||
{
|
||||
L(NoBlock);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
|
||||
union PtrCast {
|
||||
ClassPtrType ClassPtr;
|
||||
uintptr_t Data;
|
||||
};
|
||||
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
PtrCast Ptr;
|
||||
Ptr.ClassPtr = &FEXCore::Context::Context::CompileBlock;
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, reinterpret_cast<uint64_t>(CTX));
|
||||
mov(rsi, STATE);
|
||||
mov(rax, GetCompileBlockPtr());
|
||||
mov(rax, Ptr.Data);
|
||||
|
||||
call(rax);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
|
||||
// rdx already contains RIP here
|
||||
jmp(LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = getCurr<uint64_t>();
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, config.ExitFunctionLinkThis);
|
||||
mov(rsi, STATE);
|
||||
mov(rdx, rax); // rax is set at the block end
|
||||
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rax, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi}
|
||||
mov(rdi, STATE);
|
||||
mov(rsi, rax); // rax is set at the block end
|
||||
|
||||
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
jmp(r9);
|
||||
}
|
||||
else {
|
||||
jmp(rax);
|
||||
}
|
||||
mov(rax, config.ExitFunctionLink);
|
||||
call(rax);
|
||||
jmp(rax);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -304,7 +226,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
|
||||
}
|
||||
|
||||
{
|
||||
CallbackPtr = getCurr<JITCallback>();
|
||||
CallbackPtr = getCurr<CPUBackend::JITCallback>();
|
||||
|
||||
push(rbx);
|
||||
push(rbp);
|
||||
@@ -319,7 +241,8 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
|
||||
// XXX: XMM?
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
add(qword STATE_PTR(CpuStateFrame, SignalHandlerRefCounter), 1);
|
||||
mov(rax, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
|
||||
add(dword [rax], 1);
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
@@ -327,12 +250,12 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
// Guest stack is now correctly misaligned after a regular call instruction
|
||||
sub(qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]), 16);
|
||||
mov(rbx, qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 16);
|
||||
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])]);
|
||||
mov(qword [rbx], rax);
|
||||
|
||||
// Store RIP to the context state
|
||||
mov(qword STATE_PTR(CpuStateFrame, State.rip), rsi);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], rsi);
|
||||
|
||||
// Back to the loop top now
|
||||
jmp(LoopTop);
|
||||
@@ -341,41 +264,14 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
|
||||
{
|
||||
// Signal return handler
|
||||
SignalHandlerReturnAddress = getCurr<uint64_t>();
|
||||
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGILL = getCurr<uint64_t>();
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest SIGTRAP handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGTRAP = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
int3();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest SIGSEGV handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGSEGV = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
hlt();
|
||||
}
|
||||
|
||||
{
|
||||
IntCallbackReturnAddress = getCurr<uint64_t>();
|
||||
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
|
||||
// using CallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
|
||||
// rdi = thread
|
||||
// rsi = rsp
|
||||
@@ -400,100 +296,30 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherCon
|
||||
Start = reinterpret_cast<uint64_t>(getCode());
|
||||
End = Start + getSize();
|
||||
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(Start), End-Start, Name);
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline
|
||||
static thread_local Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
|
||||
size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
emit.setNewBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.mov(rax, reinterpret_cast<uint64_t>(CTX));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
emit.je(RunBlock);
|
||||
{
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.mov(rax, GuestRIP);
|
||||
emit.mov(qword STATE_PTR(CpuStateFrame, State.rip), rax);
|
||||
|
||||
// Stop the thread
|
||||
emit.mov(rax, qword STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.jmp(rax);
|
||||
}
|
||||
|
||||
emit.L(RunBlock);
|
||||
|
||||
emit.ready();
|
||||
|
||||
return emit.getSize();
|
||||
}
|
||||
|
||||
|
||||
size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
emit.setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
Label InlineIRData;
|
||||
|
||||
emit.mov(rdi, STATE);
|
||||
emit.lea(rsi, ptr[rip + InlineIRData]);
|
||||
emit.call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
|
||||
emit.jmp(qword STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
|
||||
emit.L(InlineIRData);
|
||||
|
||||
emit.ready();
|
||||
|
||||
return emit.getSize();
|
||||
#if ENABLE_JITSYMBOLS
|
||||
std::string Name = "Dispatch_" + std::to_string(::gettid());
|
||||
CTX->Symbols.Register(Start, End-Start, Name);
|
||||
#endif
|
||||
}
|
||||
|
||||
X86Dispatcher::~X86Dispatcher() {
|
||||
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
|
||||
|
||||
}
|
||||
|
||||
#ifdef _M_X86_64
|
||||
|
||||
void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
DispatcherConfig config;
|
||||
config.ExecuteBlocksWithCall = true;
|
||||
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddress;
|
||||
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddress;
|
||||
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
Dispatcher = new X86Dispatcher(ctx, Thread, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
|
||||
(uintptr_t&)Interpreter.CallbackReturn = IntCallbackReturnAddress;
|
||||
}
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
|
||||
}
|
||||
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<X86Dispatcher>(CTX, Config);
|
||||
}
|
||||
#endif
|
||||
|
||||
}
|
||||
@@ -5,24 +5,13 @@
|
||||
#define XBYAK64
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
|
||||
|
||||
virtual ~X86Dispatcher() override;
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
+139
-329
@@ -7,31 +7,20 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
#include <sys/mman.h>
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
#include "Interface/Core/VSyscall/VSyscall.inc"
|
||||
|
||||
using namespace FEXCore::X86Tables;
|
||||
|
||||
static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool HasREX, bool HasXMM, bool HasMM, uint8_t InvalidOffset = 16) {
|
||||
using GPRArray = std::array<uint32_t, 16>;
|
||||
|
||||
static constexpr GPRArray GPRIndexes = {
|
||||
constexpr std::array<uint64_t, 16> GPRIndexes = {
|
||||
// Classical ordering?
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RCX,
|
||||
@@ -51,7 +40,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
FEXCore::X86State::REG_R15,
|
||||
};
|
||||
|
||||
static constexpr GPRArray GPR8BitHighIndexes = {
|
||||
constexpr std::array<uint64_t, 16> GPR8BitHighIndexes = {
|
||||
// Classical ordering?
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RCX,
|
||||
@@ -71,7 +60,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
FEXCore::X86State::REG_R15,
|
||||
};
|
||||
|
||||
static constexpr GPRArray XMMIndexes = {
|
||||
constexpr std::array<uint64_t, 16> XMMIndexes = {
|
||||
FEXCore::X86State::REG_XMM_0,
|
||||
FEXCore::X86State::REG_XMM_1,
|
||||
FEXCore::X86State::REG_XMM_2,
|
||||
@@ -90,7 +79,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
FEXCore::X86State::REG_XMM_15,
|
||||
};
|
||||
|
||||
static constexpr GPRArray MMIndexes = {
|
||||
constexpr std::array<uint64_t, 16> MMIndexes = {
|
||||
FEXCore::X86State::REG_MM_0,
|
||||
FEXCore::X86State::REG_MM_1,
|
||||
FEXCore::X86State::REG_MM_2,
|
||||
@@ -109,7 +98,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
FEXCore::X86State::REG_INVALID
|
||||
};
|
||||
|
||||
const GPRArray *GPRs = &GPRIndexes;
|
||||
const std::array<uint64_t, 16> *GPRs = &GPRIndexes;
|
||||
if (HasXMM) {
|
||||
GPRs = &XMMIndexes;
|
||||
}
|
||||
@@ -128,79 +117,33 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
return (*GPRs)[(REX << 3) | bits];
|
||||
}
|
||||
|
||||
static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
|
||||
using GPRArray = std::array<uint32_t, 16>;
|
||||
|
||||
static constexpr GPRArray GPRIndexes = {
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RCX,
|
||||
FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_RBX,
|
||||
FEXCore::X86State::REG_RSP,
|
||||
FEXCore::X86State::REG_RBP,
|
||||
FEXCore::X86State::REG_RSI,
|
||||
FEXCore::X86State::REG_RDI,
|
||||
FEXCore::X86State::REG_R8,
|
||||
FEXCore::X86State::REG_R9,
|
||||
FEXCore::X86State::REG_R10,
|
||||
FEXCore::X86State::REG_R11,
|
||||
FEXCore::X86State::REG_R12,
|
||||
FEXCore::X86State::REG_R13,
|
||||
FEXCore::X86State::REG_R14,
|
||||
FEXCore::X86State::REG_R15,
|
||||
};
|
||||
|
||||
static constexpr GPRArray XMMIndexes = {
|
||||
FEXCore::X86State::REG_XMM_0,
|
||||
FEXCore::X86State::REG_XMM_1,
|
||||
FEXCore::X86State::REG_XMM_2,
|
||||
FEXCore::X86State::REG_XMM_3,
|
||||
FEXCore::X86State::REG_XMM_4,
|
||||
FEXCore::X86State::REG_XMM_5,
|
||||
FEXCore::X86State::REG_XMM_6,
|
||||
FEXCore::X86State::REG_XMM_7,
|
||||
FEXCore::X86State::REG_XMM_8,
|
||||
FEXCore::X86State::REG_XMM_9,
|
||||
FEXCore::X86State::REG_XMM_10,
|
||||
FEXCore::X86State::REG_XMM_11,
|
||||
FEXCore::X86State::REG_XMM_12,
|
||||
FEXCore::X86State::REG_XMM_13,
|
||||
FEXCore::X86State::REG_XMM_14,
|
||||
FEXCore::X86State::REG_XMM_15,
|
||||
};
|
||||
|
||||
if (HasXMM) {
|
||||
return XMMIndexes[vvvv];
|
||||
} else {
|
||||
return GPRIndexes[vvvv];
|
||||
}
|
||||
}
|
||||
|
||||
Decoder::Decoder(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx}
|
||||
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN }
|
||||
, PoolObject {ctx->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
}
|
||||
|
||||
Decoder::~Decoder() {
|
||||
PoolObject.UnclaimBuffer();
|
||||
: CTX {ctx} {
|
||||
DecodedBuffer.resize(DefaultDecodedBufferSize);
|
||||
}
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
uint8_t Byte = InstStream[InstructionSize];
|
||||
LOGMAN_THROW_AA_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
LogMan::Throw::A(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
Instruction[InstructionSize] = Byte;
|
||||
InstructionSize++;
|
||||
return Byte;
|
||||
}
|
||||
|
||||
uint8_t Decoder::PeekByte(uint8_t Offset) const {
|
||||
uint8_t Decoder::PeekByte(uint8_t Offset) {
|
||||
uint8_t Byte = InstStream[InstructionSize + Offset];
|
||||
return Byte;
|
||||
}
|
||||
|
||||
uint64_t Decoder::ReadData(uint8_t Size) {
|
||||
LOGMAN_THROW_AA_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
if (Size == 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (Size > sizeof(uint64_t)) {
|
||||
LogMan::Msg::A("Unknown data size to read");
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t Res = 0;
|
||||
std::memcpy(&Res, &InstStream[InstructionSize], Size);
|
||||
@@ -253,9 +196,9 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
}
|
||||
}
|
||||
|
||||
Operand->Type = DecodedOperand::OpType::SIB;
|
||||
Operand->Data.SIB.Scale = 1;
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
Operand->TypeSIB.Type = DecodedOperand::TYPE_SIB;
|
||||
Operand->TypeSIB.Scale = 1;
|
||||
Operand->TypeSIB.Offset = Literal;
|
||||
|
||||
// Only called when ModRM.mod != 0b11
|
||||
struct Encodings {
|
||||
@@ -268,34 +211,34 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RDI},
|
||||
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RSI},
|
||||
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RDI},
|
||||
{FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_INVALID, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RSI, 255},
|
||||
{FEXCore::X86State::REG_RDI, 255},
|
||||
{255, 255},
|
||||
{FEXCore::X86State::REG_RBX, 255},
|
||||
// Mod = 0b01
|
||||
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RSI},
|
||||
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RDI},
|
||||
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RSI},
|
||||
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RDI},
|
||||
{FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RSI, 255},
|
||||
{FEXCore::X86State::REG_RDI, 255},
|
||||
{FEXCore::X86State::REG_RBP, 255},
|
||||
{FEXCore::X86State::REG_RBX, 255},
|
||||
// Mod = 0b10
|
||||
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RSI},
|
||||
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RDI},
|
||||
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RSI},
|
||||
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RDI},
|
||||
{FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_INVALID},
|
||||
{FEXCore::X86State::REG_RSI, 255},
|
||||
{FEXCore::X86State::REG_RDI, 255},
|
||||
{FEXCore::X86State::REG_RBP, 255},
|
||||
{FEXCore::X86State::REG_RBX, 255},
|
||||
}};
|
||||
|
||||
uint8_t LookupIndex = ModRM.mod << 3 | ModRM.rm;
|
||||
auto it = Lookup[LookupIndex];
|
||||
Operand->Data.SIB.Base = it.Base;
|
||||
Operand->Data.SIB.Index = it.Index;
|
||||
Operand->TypeSIB.Base = it.Base;
|
||||
Operand->TypeSIB.Index = it.Index;
|
||||
}
|
||||
|
||||
void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM) {
|
||||
@@ -333,79 +276,79 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
}
|
||||
|
||||
// SIB
|
||||
Operand->Type = DecodedOperand::OpType::SIB;
|
||||
Operand->Data.SIB.Scale = 1 << SIB.scale;
|
||||
Operand->TypeSIB.Type = DecodedOperand::TYPE_SIB;
|
||||
Operand->TypeSIB.Scale = 1 << SIB.scale;
|
||||
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
Operand->TypeSIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->TypeSIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
uint64_t Literal {0};
|
||||
LogMan::Throw::A(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
if (Displacement) {
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
Operand->TypeSIB.Offset = Literal;
|
||||
}
|
||||
else if (ModRM.mod == 0) {
|
||||
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
|
||||
if (ModRM.rm == 0b101) {
|
||||
// 32bit Displacement
|
||||
const uint32_t Literal = ReadData(4);
|
||||
uint32_t Literal;
|
||||
Literal = ReadData(4);
|
||||
|
||||
Operand->Type = DecodedOperand::OpType::RIPRelative;
|
||||
Operand->Data.RIPLiteral.Value.u = Literal;
|
||||
Operand->TypeRIPLiteral.Type = DecodedOperand::TYPE_RIP_RELATIVE;
|
||||
Operand->TypeRIPLiteral.Literal.u = Literal;
|
||||
}
|
||||
else {
|
||||
// Register-direct addressing
|
||||
Operand->Type = DecodedOperand::OpType::GPRDirect;
|
||||
Operand->Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
Operand->TypeGPR.Type = DecodedOperand::TYPE_GPR_DIRECT;
|
||||
Operand->TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint8_t DisplacementSize = ModRM.mod == 1 ? 1 : 4;
|
||||
uint32_t Literal = ReadData(DisplacementSize);
|
||||
uint32_t Literal{};
|
||||
Literal = ReadData(DisplacementSize);
|
||||
if (DisplacementSize == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
Displacement = DisplacementSize;
|
||||
|
||||
Operand->Type = DecodedOperand::OpType::GPRIndirect;
|
||||
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
Operand->Data.GPRIndirect.Displacement = Literal;
|
||||
Operand->TypeGPRIndirect.Type = DecodedOperand::TYPE_GPR_INDIRECT;
|
||||
Operand->TypeGPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
Operand->TypeGPRIndirect.Displacement = Literal;
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options) {
|
||||
bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op) {
|
||||
DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
// XXX: Once we support 32bit x86 then this will be necessary to support
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
|
||||
LogMan::Msg::DFmt("Legacy Prefix");
|
||||
LogMan::Msg::D("Legacy Prefix");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
"Group Ops should have been decoded before this!");
|
||||
LogMan::Throw::A(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
"Group Ops should have been decoded before this!");
|
||||
|
||||
uint8_t DestSize{};
|
||||
const bool HasWideningDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST) != 0 ||
|
||||
(Options.w && CTX->Config.Is64BitMode);
|
||||
const bool HasNarrowingDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST) != 0;
|
||||
bool HasWideningDisplacement = FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST;
|
||||
bool HasNarrowingDisplacement = FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST;
|
||||
|
||||
bool HasXMMSrc = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
@@ -438,8 +381,8 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
// New instruction size decoding
|
||||
{
|
||||
// Decode destinations first
|
||||
const auto DstSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeDstFlags(Info->Flags);
|
||||
const auto SrcSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeSrcFlags(Info->Flags);
|
||||
uint32_t DstSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeDstFlags(Info->Flags);
|
||||
uint32_t SrcSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeSrcFlags(Info->Flags);
|
||||
|
||||
if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_8BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_8BIT);
|
||||
@@ -516,25 +459,22 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ||
|
||||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RDX)) {
|
||||
// Some instructions hardcode their destination as RAX
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR = HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
|
||||
CurrentDest->TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
CurrentDest->TypeGPR.HighBits = false;
|
||||
CurrentDest->TypeGPR.GPR = HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
|
||||
CurrentDest = &DecodeInst->Src[0];
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_AA_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
LogMan::Throw::A(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
// This also means that the destination is always a GPR on these ones
|
||||
// ADDITIONALLY:
|
||||
// If there is a REX prefix then that allows extended GPR usage
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Dest.Data.GPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100) || HasHighXMM;
|
||||
CurrentDest->Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
|
||||
|
||||
if (CurrentDest->Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
return false;
|
||||
CurrentDest->TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
DecodeInst->Dest.TypeGPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100) || HasHighXMM;
|
||||
CurrentDest->TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
|
||||
}
|
||||
|
||||
uint8_t Bytes = Info->MoreBytes;
|
||||
@@ -557,86 +497,55 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
|
||||
// Decode the GPR source first
|
||||
GPR.Type = DecodedOperand::OpType::GPR;
|
||||
GPR.Data.GPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX) || HasHighXMM;
|
||||
GPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_R ? 1 : 0, ModRM.reg, GPR8Bit, HasREX, HasXMMGPR, HasMMGPR);
|
||||
|
||||
if (GPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
return false;
|
||||
GPR.TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
GPR.TypeGPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX) || HasHighXMM;
|
||||
GPR.TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_R ? 1 : 0, ModRM.reg, GPR8Bit, HasREX, HasXMMGPR, HasMMGPR);
|
||||
|
||||
// ModRM.mod == 0b11 == Register
|
||||
// ModRM.Mod != 0b11 == Register-direct addressing
|
||||
if (ModRM.mod == 0b11) {
|
||||
NonGPR.Type = DecodedOperand::OpType::GPR;
|
||||
NonGPR.Data.GPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX) || HasHighXMM;
|
||||
NonGPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, NonGPR8Bit, HasREX, HasXMMNonGPR, HasMMNonGPR);
|
||||
if (NonGPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
return false;
|
||||
NonGPR.TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
NonGPR.TypeGPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX) || HasHighXMM;
|
||||
NonGPR.TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, NonGPR8Bit, HasREX, HasXMMNonGPR, HasMMNonGPR);
|
||||
}
|
||||
else {
|
||||
// Only decode if we haven't pre-decoded
|
||||
if (NonGPR.IsNone()) {
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&NonGPR, ModRM);
|
||||
}
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&NonGPR, ModRM);
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
size_t CurrentSrc = 0;
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) != 0) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM) {
|
||||
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SF_MOD_DST) {
|
||||
if (!ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest))
|
||||
return false;
|
||||
ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest);
|
||||
}
|
||||
else {
|
||||
if (!ModRMOperand(DecodeInst->Dest, DecodeInst->Src[CurrentSrc], HasXMMDst, HasXMMSrc, HasMMDst, HasMMSrc, Is8BitDest, Is8BitSrc))
|
||||
return false;
|
||||
ModRMOperand(DecodeInst->Dest, DecodeInst->Src[CurrentSrc], HasXMMDst, HasXMMSrc, HasMMDst, HasMMSrc, Is8BitDest, Is8BitSrc);
|
||||
}
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) != 0) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RAX)) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RAX;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.GPR = FEXCore::X86State::REG_RAX;
|
||||
++CurrentSrc;
|
||||
}
|
||||
else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RCX)) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RCX;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.GPR = FEXCore::X86State::REG_RCX;
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) != 0) {
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_AA_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
LogMan::Throw::A(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
DecodeInst->Src[CurrentSrc].TypeLiteral.Size = Bytes;
|
||||
|
||||
uint64_t Literal = ReadData(Bytes);
|
||||
uint64_t Literal {0};
|
||||
Literal = ReadData(Bytes);
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT) ||
|
||||
(DecodeFlags::GetSizeDstFlags(DecodeInst->Flags) == DecodeFlags::SIZE_64BIT && Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT64BIT)) {
|
||||
@@ -649,16 +558,15 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
else {
|
||||
Literal = static_cast<int32_t>(Literal);
|
||||
}
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
|
||||
DecodeInst->Src[CurrentSrc].TypeLiteral.Size = DestSize;
|
||||
}
|
||||
|
||||
Bytes = 0;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
DecodeInst->Src[CurrentSrc].TypeLiteral.Type = DecodedOperand::TYPE_LITERAL;
|
||||
DecodeInst->Src[CurrentSrc].TypeLiteral.Literal = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
|
||||
DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
LogMan::Throw::A(Bytes == 0, "Inst at 0x%lx: 0x%04x '%s' Had an instruction of size %d with %d remaining", DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name, InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return true;
|
||||
}
|
||||
@@ -669,22 +577,21 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
|
||||
// XXX: Once we support 32bit x86 then this will be necessary to support
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
|
||||
LogMan::Msg::DFmt("Legacy Prefix");
|
||||
LogMan::Msg::D("Legacy Prefix");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
|
||||
"REX PREFIX should have been decoded before this!");
|
||||
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
|
||||
|
||||
if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
|
||||
Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
|
||||
@@ -740,7 +647,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
3,
|
||||
};
|
||||
uint8_t Field = RegToField[ModRM.reg];
|
||||
LOGMAN_THROW_AA_FMT(Field != 255, "Invalid field selected!");
|
||||
LogMan::Throw::A(Field != 255, "Invalid field selected!");
|
||||
|
||||
LocalOp = (Field << 3) | ModRM.rm;
|
||||
return NormalOp(&SecondModRMTableOps[LocalOp], LocalOp);
|
||||
@@ -762,38 +669,19 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
return NormalOp(&X87Ops[X87Op], X87Op);
|
||||
}
|
||||
else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
|
||||
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
|
||||
uint16_t map_select = 1;
|
||||
uint16_t pp = 0;
|
||||
const uint8_t Byte1 = ReadByte();
|
||||
DecodedHeader options{};
|
||||
|
||||
if ((Byte1 & 0b10000000) == 0) {
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.R shouldn't be 0 in 32-bit mode!");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
|
||||
}
|
||||
uint8_t Byte1 = ReadByte();
|
||||
|
||||
if (Op == 0xC5) { // Two byte VEX
|
||||
pp = Byte1 & 0b11;
|
||||
options.vvvv = 15 - ((Byte1 & 0b01111000) >> 3);
|
||||
}
|
||||
else { // 0xC4 = Three byte VEX
|
||||
const uint8_t Byte2 = ReadByte();
|
||||
uint8_t Byte2 = ReadByte();
|
||||
pp = Byte2 & 0b11;
|
||||
map_select = Byte1 & 0b11111;
|
||||
options.vvvv = 15 - ((Byte2 & 0b01111000) >> 3);
|
||||
options.w = (Byte2 & 0b10000000) != 0;
|
||||
if ((Byte1 & 0b01000000) == 0) {
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.X shouldn't be 0 in 32-bit mode!");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
}
|
||||
if (CTX->Config.Is64BitMode && (Byte1 & 0b00100000) == 0) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
|
||||
}
|
||||
if (!(map_select >= 1 && map_select <= 3)) {
|
||||
LogMan::Msg::EFmt("We don't understand a map_select of: {}", map_select);
|
||||
return false;
|
||||
}
|
||||
LogMan::Throw::A(map_select >= 1 && map_select <= 3, "We don't understand a map_select of: %d", map_select);
|
||||
}
|
||||
|
||||
uint16_t VEXOp = ReadByte();
|
||||
@@ -805,7 +693,6 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
|
||||
if (LocalInfo->Type >= FEXCore::X86Tables::TYPE_VEX_GROUP_12 &&
|
||||
LocalInfo->Type <= FEXCore::X86Tables::TYPE_VEX_GROUP_17) {
|
||||
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
@@ -817,14 +704,12 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
|
||||
Op = OPD(LocalInfo->Type, pp, ModRM.reg);
|
||||
#undef OPD
|
||||
return NormalOp(&VEXTableGroupOps[Op], Op, options);
|
||||
} else {
|
||||
return NormalOp(LocalInfo, Op, options);
|
||||
return NormalOp(&VEXTableGroupOps[Op], Op);
|
||||
}
|
||||
else
|
||||
return NormalOp(LocalInfo, Op);
|
||||
}
|
||||
else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
|
||||
|
||||
/* uint8_t P1 = */ ReadByte();
|
||||
/* uint8_t P2 = */ ReadByte();
|
||||
/* uint8_t P3 = */ ReadByte();
|
||||
@@ -844,14 +729,12 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
DecodeInst->PC = PC;
|
||||
|
||||
for(;;) {
|
||||
if (InstructionSize >= MAX_INST_SIZE)
|
||||
return false;
|
||||
uint8_t Op = ReadByte();
|
||||
switch (Op) {
|
||||
case 0x0F: {// Escape Op
|
||||
uint8_t EscapeOp = ReadByte();
|
||||
switch (EscapeOp) {
|
||||
case 0x0F: [[unlikely]] { // 3DNow!
|
||||
case 0x0F: { // 3DNow!
|
||||
// 3DNow! Instruction Encoding: 0F 0F [ModRM] [SIB] [Displacement] [Opcode]
|
||||
// Decode ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
@@ -864,12 +747,8 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
const bool Has16BitAddressing = !CTX->Config.Is64BitMode &&
|
||||
DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
|
||||
|
||||
// All 3DNow! instructions have the second argument as the rm handler
|
||||
// We need to decode it upfront to get the displacement out of the way
|
||||
if (ModRM.mod != 0b11) {
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&DecodeInst->Src[0], ModRM);
|
||||
}
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&DecodeInst->Src[0], ModRM);
|
||||
|
||||
// Take a peek at the op just past the displacement
|
||||
uint8_t LocalOp = ReadByte();
|
||||
@@ -878,20 +757,14 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
case 0x38: { // F38 Table!
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
constexpr uint16_t PF_38_66 = 1;
|
||||
constexpr uint16_t PF_38_F2 = 2;
|
||||
|
||||
uint16_t Prefix = PF_38_NONE;
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_OPERAND_SIZE) {
|
||||
Prefix |= PF_38_66;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REPNE_PREFIX) {
|
||||
Prefix |= PF_38_F2;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REP_PREFIX) {
|
||||
Prefix |= PF_38_F3;
|
||||
}
|
||||
if (DecodeInst->LastEscapePrefix == 0xF2) // REPNE
|
||||
Prefix = PF_38_F2;
|
||||
else if (DecodeInst->LastEscapePrefix == 0x66) // Operand Size
|
||||
Prefix = PF_38_66;
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
return NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp);
|
||||
@@ -1006,7 +879,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
auto Info = &FEXCore::X86Tables::BaseOps[Op];
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
|
||||
LogMan::Throw::A(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
|
||||
|
||||
// Widening displacement
|
||||
@@ -1036,10 +909,6 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
assert(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -1049,7 +918,7 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
|
||||
// If the RIP setting is conditional AND within our symbol range then it can be considered for multiblock
|
||||
uint64_t TargetRIP = 0;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
|
||||
bool Conditional = true;
|
||||
|
||||
switch (DecodeInst->OP) {
|
||||
@@ -1059,23 +928,19 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// auto RIPOffset = LoadSource(Op, Op->Src[0], Op->Flags);
|
||||
// auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
|
||||
// Target offset is PC + InstSize + Literal
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
LogMan::Throw::A(DecodeInst->Src[0].TypeNone.Type == DecodedOperand::TYPE_LITERAL, "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].TypeLiteral.Literal;
|
||||
break;
|
||||
}
|
||||
case 0xE9:
|
||||
case 0xEB: // Both are unconditional JMP instructions
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
LogMan::Throw::A(DecodeInst->Src[0].TypeNone.Type == DecodedOperand::TYPE_LITERAL, "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].TypeLiteral.Literal;
|
||||
Conditional = false;
|
||||
break;
|
||||
case 0xE8: // Call - Immediate target, We don't want to inline calls
|
||||
if (ExternalBranches) {
|
||||
ExternalBranches->insert(DecodeInst->PC + DecodeInst->InstSize);
|
||||
}
|
||||
[[fallthrough]];
|
||||
case 0xC2: // RET imm
|
||||
case 0xC3: // RET
|
||||
case 0xE8: // Call - Immediate target, We don't want to inline calls
|
||||
default:
|
||||
return;
|
||||
break;
|
||||
@@ -1105,33 +970,10 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
BlocksToDecode.find(TargetRIP) == BlocksToDecode.end()) {
|
||||
BlocksToDecode.emplace(TargetRIP);
|
||||
}
|
||||
} else {
|
||||
if (ExternalBranches) {
|
||||
ExternalBranches->insert(TargetRIP);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
|
||||
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
|
||||
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
|
||||
|
||||
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64 &&
|
||||
RIP >= VSyscall_Base &&
|
||||
RIP < VSyscall_End) {
|
||||
// VSyscall
|
||||
// This doesn't exist on AArch64 and on x86_64 hosts this is emulated with faults to a region mapped with --xp permissions
|
||||
// Offset 0: vgettimeofday
|
||||
// Offset 0x400: vtime
|
||||
// Offset 0x800: vgetcpu
|
||||
uint64_t Offset = RIP - VSyscall_Base;
|
||||
return VSyscallData + Offset;
|
||||
}
|
||||
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC) {
|
||||
Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
@@ -1139,19 +981,19 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
DecodedSize = 0;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// XXX: Load symbol data
|
||||
SymbolAvailable = false;
|
||||
EntryPoint = PC;
|
||||
InstStream = _InstStream;
|
||||
|
||||
bool ErrorDuringDecoding = false;
|
||||
uint64_t TotalInstructions{};
|
||||
|
||||
// If we don't have symbols available then we become a bit optimistic about multiblock ranges
|
||||
if (!SymbolAvailable) {
|
||||
// If we don't have a symbol available then assume all branches are valid for multiblock
|
||||
SymbolMaxAddress = SectionMaxAddress;
|
||||
SymbolMaxAddress = ~0ULL;
|
||||
SymbolMinAddress = EntryPoint;
|
||||
}
|
||||
|
||||
@@ -1161,13 +1003,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
// Entry is a jump target
|
||||
BlocksToDecode.emplace(PC);
|
||||
|
||||
uint64_t CurrentCodePage = PC & FHU::FEX_PAGE_MASK;
|
||||
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1181,41 +1016,20 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
uint64_t BlockStartOffset = DecodedSize;
|
||||
|
||||
// Do a bit of pointer math to figure out where we are in code
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
InstStream = _InstStream - EntryPoint + RIPToDecode;
|
||||
|
||||
while (1) {
|
||||
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpMinAddress = RIPToDecode + PCOffset;
|
||||
auto OpMaxAddress = OpMinAddress + MAX_INST_SIZE;
|
||||
|
||||
auto OpMinPage = OpMinAddress & FHU::FEX_PAGE_MASK;
|
||||
auto OpMaxPage = OpMaxAddress & FHU::FEX_PAGE_MASK;
|
||||
|
||||
|
||||
if (OpMinPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMinPage;
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpMaxPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMaxPage;
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) {
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", RIPToDecode + PCOffset, PC);
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
LogMan::Msg::D("Couldn't Decode something at 0x%lx, Started at 0x%lx", PC + PCOffset, PC);
|
||||
LogMan::Throw::A(Blocks.size() != 1, "Decode Error in entry block");
|
||||
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
DecodeInst->InstSize = 0;
|
||||
if (ErrorDuringDecoding && Blocks.size() != 1) {
|
||||
ErrorDuringDecoding = false;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
|
||||
@@ -1224,11 +1038,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
++BlockNumberOfInstructions;
|
||||
++DecodedSize;
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) {
|
||||
break;
|
||||
}
|
||||
|
||||
bool CanContinue = false;
|
||||
if (!(DecodeInst->TableInfo->Flags &
|
||||
(FEXCore::X86Tables::InstFlags::FLAGS_BLOCK_END | FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP))) {
|
||||
@@ -1248,7 +1057,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
}
|
||||
|
||||
if (DecodedSize >= CTX->Config.MaxInstPerBlock ||
|
||||
DecodedSize >= DefaultDecodedBufferSize) {
|
||||
DecodedSize >= DecodedBuffer.size()) {
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1265,7 +1074,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
|
||||
// Copy over only the number of instructions we decoded
|
||||
CurrentBlockDecoding.NumInstructions = BlockNumberOfInstructions;
|
||||
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer.at(BlockStartOffset);
|
||||
}
|
||||
|
||||
|
||||
@@ -1273,6 +1082,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
std::sort(Blocks.begin(), Blocks.end(), [](const FEXCore::Frontend::Decoder::DecodedBlocks& a, const FEXCore::Frontend::Decoder::DecodedBlocks& b) {
|
||||
return a.Entry < b.Entry;
|
||||
});
|
||||
return !ErrorDuringDecoding;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+9
-36
@@ -1,13 +1,11 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <set>
|
||||
#include <stddef.h>
|
||||
#include <stack>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
@@ -26,49 +24,31 @@ public:
|
||||
};
|
||||
|
||||
Decoder(FEXCore::Context::Context *ctx);
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
bool DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
|
||||
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() {
|
||||
return &Blocks;
|
||||
}
|
||||
|
||||
uint64_t DecodedMinAddress {};
|
||||
uint64_t DecodedMaxAddress {~0ULL};
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
|
||||
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
|
||||
private:
|
||||
// To pass any information from instruction prefixes
|
||||
// down into the actual instruction handling machinery.
|
||||
struct DecodedHeader {
|
||||
uint8_t vvvv; // Encoded operand in a VEX prefix.
|
||||
bool w; // VEX.W bit.
|
||||
};
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
const FEXCore::HLE::SyscallOSABI OSABI{};
|
||||
|
||||
bool DecodeInstruction(uint64_t PC);
|
||||
|
||||
void BranchTargetInMultiblockRange();
|
||||
|
||||
uint8_t ReadByte();
|
||||
uint8_t PeekByte(uint8_t Offset) const;
|
||||
uint8_t PeekByte(uint8_t Offset);
|
||||
uint64_t ReadData(uint8_t Size);
|
||||
void SkipBytes(uint8_t Size) { InstructionSize += Size; }
|
||||
|
||||
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options = {});
|
||||
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
|
||||
bool NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
|
||||
|
||||
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
|
||||
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
std::vector<FEXCore::X86Tables::DecodedInst> DecodedBuffer;
|
||||
size_t DecodedSize {};
|
||||
|
||||
uint8_t const *InstStream;
|
||||
@@ -85,26 +65,19 @@ private:
|
||||
uint64_t MaxCondBranchBackwards {~0ULL};
|
||||
uint64_t SymbolMaxAddress {};
|
||||
uint64_t SymbolMinAddress {~0ULL};
|
||||
uint64_t SectionMaxAddress {~0ULL};
|
||||
|
||||
std::vector<DecodedBlocks> Blocks;
|
||||
std::set<uint64_t> BlocksToDecode;
|
||||
std::set<uint64_t> HasBlocks;
|
||||
std::set<uint64_t> *ExternalBranches {nullptr};
|
||||
|
||||
// ModRM rm decoding
|
||||
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
void DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
void DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
|
||||
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp{
|
||||
const std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp {
|
||||
&FEXCore::Frontend::Decoder::DecodeModRM_64,
|
||||
&FEXCore::Frontend::Decoder::DecodeModRM_16,
|
||||
};
|
||||
|
||||
const uint8_t *AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
|
||||
FEXCORE_TELEMETRY_INIT(VEXOpTelem, TYPE_USES_VEX_OPS);
|
||||
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
|
||||
};
|
||||
}
|
||||
+518
-827
File diff suppressed because it is too large.
Load diff
+16
-40
@@ -5,23 +5,18 @@ $end_info$
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <mutex>
|
||||
#include <thread>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Common/NetStream.h"
|
||||
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
class GdbServer {
|
||||
public:
|
||||
GdbServer(FEXCore::Context::Context *ctx);
|
||||
@@ -29,24 +24,16 @@ public:
|
||||
// Public for threading
|
||||
void GdbServerLoop();
|
||||
|
||||
void AlertLibrariesChanged() {
|
||||
LibraryMapChanged = true;
|
||||
}
|
||||
|
||||
private:
|
||||
void Break(int signal);
|
||||
|
||||
void OpenListenSocket();
|
||||
std::unique_ptr<std::iostream> OpenSocket();
|
||||
void StartThread();
|
||||
std::string ReadPacket(std::iostream &stream);
|
||||
void SendPacket(std::ostream &stream, const std::string& packet);
|
||||
void SendPacket(std::ostream &stream, std::string packet);
|
||||
|
||||
void SendACK(std::ostream &stream, bool NACK);
|
||||
|
||||
Event ThreadBreakEvent{};
|
||||
void WaitForThreadWakeup();
|
||||
|
||||
struct HandledPacketType {
|
||||
std::string Response{};
|
||||
enum ResponseType {
|
||||
@@ -60,20 +47,18 @@ private:
|
||||
ResponseType TypeResponse{};
|
||||
};
|
||||
|
||||
void SendPacketPair(const HandledPacketType& packetPair);
|
||||
HandledPacketType ProcessPacket(const std::string &packet);
|
||||
HandledPacketType handleQuery(const std::string &packet);
|
||||
HandledPacketType handleXfer(const std::string &packet);
|
||||
HandledPacketType handleMemory(const std::string &packet);
|
||||
HandledPacketType handleV(const std::string& packet);
|
||||
HandledPacketType handleThreadOp(const std::string &packet);
|
||||
HandledPacketType handleBreakpoint(const std::string &packet);
|
||||
void SendPacketPair(HandledPacketType packetPair);
|
||||
HandledPacketType ProcessPacket(std::string &packet);
|
||||
HandledPacketType handleQuery(std::string &packet);
|
||||
HandledPacketType handleXfer(std::string &packet);
|
||||
HandledPacketType handleMemory(std::string &packet);
|
||||
HandledPacketType handleV(std::string& packet);
|
||||
HandledPacketType handleThreadOp(std::string &packet);
|
||||
HandledPacketType handleBreakpoint(std::string &packet);
|
||||
HandledPacketType handleProgramOffsets();
|
||||
|
||||
HandledPacketType ThreadAction(char action, uint32_t tid);
|
||||
|
||||
std::string readRegs();
|
||||
HandledPacketType readReg(const std::string& packet);
|
||||
HandledPacketType readReg(std::string& packet);
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
std::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
@@ -81,17 +66,8 @@ private:
|
||||
std::mutex sendMutex;
|
||||
bool SettingNoAckMode{false};
|
||||
bool NoAckMode{false};
|
||||
bool NonStopMode{false};
|
||||
std::string ThreadString{};
|
||||
std::string OSDataString{};
|
||||
void buildLibraryMap();
|
||||
std::atomic<bool> LibraryMapChanged = true;
|
||||
std::string LibraryMapString{};
|
||||
|
||||
// Used to keep track of which signals to pass to the guest
|
||||
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
|
||||
uint32_t CurrentDebuggingThread{};
|
||||
int ListenSocket{};
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
};
|
||||
|
||||
|
||||
+1
-118
@@ -1,5 +1,4 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
@@ -14,130 +13,14 @@
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
// Data Zero Prohibited flag
|
||||
// 0b0 = ZVA/GVA/GZVA permitted
|
||||
// 0b1 = ZVA/GVA/GZVA prohibited
|
||||
constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
|
||||
// Log2 of the blocksize in 32-bit words
|
||||
constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
static uint32_t GetDCZID() {
|
||||
uint64_t Result{};
|
||||
__asm("mrs %[Res], DCZID_EL0"
|
||||
: [Res] "=r" (Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static uint32_t GetFPCR() {
|
||||
uint64_t Result{};
|
||||
__asm ("mrs %[Res], FPCR"
|
||||
: [Res] "=r" (Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static void SetFPCR(uint64_t Value) {
|
||||
__asm ("msr FPCR, %[Value]"
|
||||
:: [Value] "r" (Value));
|
||||
}
|
||||
|
||||
#else
|
||||
static uint32_t GetDCZID() {
|
||||
// Return unsupported
|
||||
return DCZID_DZP_MASK;
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef _M_ARM_64
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
|
||||
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
|
||||
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
__asm volatile ("mrs %[ctr], ctr_el0"
|
||||
: [ctr] "=r"(CTR));
|
||||
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
#endif
|
||||
#ifdef _M_X86_64
|
||||
Xbyak::util::Cpu Features{};
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
Supports3DNow = Features.has(Xbyak::util::Cpu::t3DN) && Features.has(Xbyak::util::Cpu::tE3DN);
|
||||
SupportsSSE4A = Features.has(Xbyak::util::Cpu::tSSE4a);
|
||||
SupportsAVX = true;
|
||||
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
|
||||
SupportsCLZERO = ebx & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#else
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps =
|
||||
(1U << 8) | // Invalid Operation float exception trap enable
|
||||
(1U << 9) | // Divide by zero float exception trap enable
|
||||
(1U << 10) | // Overflow float exception trap enable
|
||||
(1U << 11) | // Underflow float exception trap enable
|
||||
(1U << 12) | // Inexact float exception trap enable
|
||||
(1U << 15); // Input Denormal float exception trap enable
|
||||
|
||||
uint32_t OriginalFPCR = GetFPCR();
|
||||
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
|
||||
SetFPCR(FPCR);
|
||||
FPCR = GetFPCR();
|
||||
SupportsFloatExceptions = (FPCR & ExceptionEnableTraps) == ExceptionEnableTraps;
|
||||
|
||||
// Set FPCR back to original just in case anything changed
|
||||
SetFPCR(OriginalFPCR);
|
||||
#endif
|
||||
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
if ((DCZID & DCZID_DZP_MASK) == 0) {
|
||||
uint32_t DCZID_Log2 = DCZID & DCZID_BS_MASK;
|
||||
uint32_t DCZID_Bytes = (1 << DCZID_Log2) * sizeof(uint32_t);
|
||||
// If the DC ZVA size matches the emulated cache line size
|
||||
// This means we can use the instruction
|
||||
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
#pragma once
|
||||
|
||||
namespace FEXCore {
|
||||
class HostFeatures final {
|
||||
public:
|
||||
HostFeatures();
|
||||
bool SupportsAES{};
|
||||
};
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
@@ -1,777 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#ifdef _M_X86_64
|
||||
uint8_t AtomicFetchNeg(uint8_t *Addr) {
|
||||
using Type = uint8_t;
|
||||
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
|
||||
Type Expected = MemData->load();
|
||||
Type Desired = -Expected;
|
||||
do {
|
||||
Desired = -Expected;
|
||||
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
|
||||
|
||||
return Expected;
|
||||
}
|
||||
|
||||
uint16_t AtomicFetchNeg(uint16_t *Addr) {
|
||||
using Type = uint16_t;
|
||||
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
|
||||
Type Expected = MemData->load();
|
||||
Type Desired = -Expected;
|
||||
do {
|
||||
Desired = -Expected;
|
||||
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
|
||||
|
||||
return Expected;
|
||||
}
|
||||
|
||||
uint32_t AtomicFetchNeg(uint32_t *Addr) {
|
||||
using Type = uint32_t;
|
||||
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
|
||||
Type Expected = MemData->load();
|
||||
Type Desired = -Expected;
|
||||
do {
|
||||
Desired = -Expected;
|
||||
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
|
||||
|
||||
return Expected;
|
||||
}
|
||||
|
||||
uint64_t AtomicFetchNeg(uint64_t *Addr) {
|
||||
using Type = uint64_t;
|
||||
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
|
||||
Type Expected = MemData->load();
|
||||
Type Desired = -Expected;
|
||||
do {
|
||||
Desired = -Expected;
|
||||
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
|
||||
|
||||
return Expected;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T AtomicCompareAndSwap(T expected, T desired, T *addr)
|
||||
{
|
||||
std::atomic<T> *MemData = reinterpret_cast<std::atomic<T>*>(addr);
|
||||
|
||||
T Src1 = expected;
|
||||
T Src2 = desired;
|
||||
|
||||
T Expected = Src1;
|
||||
bool Result = MemData->compare_exchange_strong(Expected, Src2);
|
||||
|
||||
return Result ? Src1 : Expected;
|
||||
}
|
||||
|
||||
template uint8_t AtomicCompareAndSwap<uint8_t>(uint8_t expected, uint8_t desired, uint8_t *addr);
|
||||
template uint16_t AtomicCompareAndSwap<uint16_t>(uint16_t expected, uint16_t desired, uint16_t *addr);
|
||||
template uint32_t AtomicCompareAndSwap<uint32_t>(uint32_t expected, uint32_t desired, uint32_t *addr);
|
||||
template uint64_t AtomicCompareAndSwap<uint64_t>(uint64_t expected, uint64_t desired, uint64_t *addr);
|
||||
|
||||
#else
|
||||
// Needs to match what the AArch64 JIT and unaligned signal handler expects
|
||||
uint8_t AtomicFetchNeg(uint8_t *Addr) {
|
||||
using Type = uint8_t;
|
||||
Type Result{};
|
||||
Type Tmp{};
|
||||
Type TmpStatus{};
|
||||
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxrb %w[Result], [%[Memory]];
|
||||
neg %w[Tmp], %w[Result];
|
||||
stlxrb %w[TmpStatus], %w[Tmp], [%[Memory]];
|
||||
cbnz %w[TmpStatus], 1b;
|
||||
)"
|
||||
: [Result] "=r" (Result)
|
||||
, [Tmp] "=r" (Tmp)
|
||||
, [TmpStatus] "=r" (TmpStatus)
|
||||
, [Memory] "+r" (Addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint16_t AtomicFetchNeg(uint16_t *Addr) {
|
||||
using Type = uint16_t;
|
||||
Type Result{};
|
||||
Type Tmp{};
|
||||
Type TmpStatus{};
|
||||
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxrh %w[Result], [%[Memory]];
|
||||
neg %w[Tmp], %w[Result];
|
||||
stlxrh %w[TmpStatus], %w[Tmp], [%[Memory]];
|
||||
cbnz %w[TmpStatus], 1b;
|
||||
)"
|
||||
: [Result] "=r" (Result)
|
||||
, [Tmp] "=r" (Tmp)
|
||||
, [TmpStatus] "=r" (TmpStatus)
|
||||
, [Memory] "+r" (Addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint32_t AtomicFetchNeg(uint32_t *Addr) {
|
||||
using Type = uint32_t;
|
||||
Type Result{};
|
||||
Type Tmp{};
|
||||
Type TmpStatus{};
|
||||
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxr %w[Result], [%[Memory]];
|
||||
neg %w[Tmp], %w[Result];
|
||||
stlxr %w[TmpStatus], %w[Tmp], [%[Memory]];
|
||||
cbnz %w[TmpStatus], 1b;
|
||||
)"
|
||||
: [Result] "=r" (Result)
|
||||
, [Tmp] "=r" (Tmp)
|
||||
, [TmpStatus] "=r" (TmpStatus)
|
||||
, [Memory] "+r" (Addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint64_t AtomicFetchNeg(uint64_t *Addr) {
|
||||
using Type = uint64_t;
|
||||
Type Result{};
|
||||
Type Tmp{};
|
||||
Type TmpStatus{};
|
||||
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxr %[Result], [%[Memory]];
|
||||
neg %[Tmp], %[Result];
|
||||
stlxr %w[TmpStatus], %[Tmp], [%[Memory]];
|
||||
cbnz %w[TmpStatus], 1b;
|
||||
)"
|
||||
: [Result] "=r" (Result)
|
||||
, [Tmp] "=r" (Tmp)
|
||||
, [TmpStatus] "=r" (TmpStatus)
|
||||
, [Memory] "+r" (Addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<>
|
||||
uint8_t AtomicCompareAndSwap(uint8_t expected, uint8_t desired, uint8_t *addr) {
|
||||
using Type = uint8_t;
|
||||
//force Result to r9 (scratch register) or clang spills to stack
|
||||
register Type Result asm("r9"){};
|
||||
Type Tmp{};
|
||||
Type Tmp2{};
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxrb %w[Tmp], [%[Memory]];
|
||||
cmp %w[Tmp], %w[Expected], uxtb;
|
||||
b.ne 2f;
|
||||
stlxrb %w[Tmp2], %w[Desired], [%[Memory]];
|
||||
cbnz %w[Tmp2], 1b;
|
||||
mov %w[Result], %w[Expected];
|
||||
b 3f;
|
||||
2:
|
||||
mov %w[Result], %w[Tmp];
|
||||
clrex;
|
||||
3:
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp)
|
||||
, [Tmp2] "=r" (Tmp2)
|
||||
, [Desired] "+r" (desired)
|
||||
, [Expected] "+r" (expected)
|
||||
, [Result] "=r" (Result)
|
||||
, [Memory] "+r" (addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<>
|
||||
uint16_t AtomicCompareAndSwap(uint16_t expected, uint16_t desired, uint16_t *addr) {
|
||||
using Type = uint16_t;
|
||||
//force Result to r9 (scratch register) or clang spills to stack
|
||||
register Type Result asm("r9"){};
|
||||
Type Tmp{};
|
||||
Type Tmp2{};
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxrh %w[Tmp], [%[Memory]];
|
||||
cmp %w[Tmp], %w[Expected], uxth;
|
||||
b.ne 2f;
|
||||
stlxrh %w[Tmp2], %w[Desired], [%[Memory]];
|
||||
cbnz %w[Tmp2], 1b;
|
||||
mov %w[Result], %w[Expected];
|
||||
b 3f;
|
||||
2:
|
||||
mov %w[Result], %w[Tmp];
|
||||
clrex;
|
||||
3:
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp)
|
||||
, [Tmp2] "=r" (Tmp2)
|
||||
, [Desired] "+r" (desired)
|
||||
, [Expected] "+r" (expected)
|
||||
, [Result] "=r" (Result)
|
||||
, [Memory] "+r" (addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<>
|
||||
uint32_t AtomicCompareAndSwap(uint32_t expected, uint32_t desired, uint32_t *addr) {
|
||||
using Type = uint32_t;
|
||||
//force Result to r9 (scratch register) or clang spills to stack
|
||||
register Type Result asm("r9"){};
|
||||
Type Tmp{};
|
||||
Type Tmp2{};
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxr %w[Tmp], [%[Memory]];
|
||||
cmp %w[Tmp], %w[Expected];
|
||||
b.ne 2f;
|
||||
stlxr %w[Tmp2], %w[Desired], [%[Memory]];
|
||||
cbnz %w[Tmp2], 1b;
|
||||
mov %w[Result], %w[Expected];
|
||||
b 3f;
|
||||
2:
|
||||
mov %w[Result], %w[Tmp];
|
||||
clrex;
|
||||
3:
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp)
|
||||
, [Tmp2] "=r" (Tmp2)
|
||||
, [Desired] "+r" (desired)
|
||||
, [Expected] "+r" (expected)
|
||||
, [Result] "=r" (Result)
|
||||
, [Memory] "+r" (addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<>
|
||||
uint64_t AtomicCompareAndSwap(uint64_t expected, uint64_t desired, uint64_t *addr) {
|
||||
using Type = uint64_t;
|
||||
//force Result to r9 (scratch register) or clang spills to stack
|
||||
register Type Result asm("r9"){};
|
||||
Type Tmp{};
|
||||
Type Tmp2{};
|
||||
__asm__ volatile(
|
||||
R"(
|
||||
1:
|
||||
ldaxr %[Tmp], [%[Memory]];
|
||||
cmp %[Tmp], %[Expected];
|
||||
b.ne 2f;
|
||||
stlxr %w[Tmp2], %[Desired], [%[Memory]];
|
||||
cbnz %w[Tmp2], 1b;
|
||||
mov %[Result], %[Expected];
|
||||
b 3f;
|
||||
2:
|
||||
mov %[Result], %[Tmp];
|
||||
clrex;
|
||||
3:
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp)
|
||||
, [Tmp2] "=r" (Tmp2)
|
||||
, [Desired] "+r" (desired)
|
||||
, [Expected] "+r" (expected)
|
||||
, [Result] "=r" (Result)
|
||||
, [Memory] "+r" (addr)
|
||||
:: "memory"
|
||||
);
|
||||
return Result;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
|
||||
// Size is the size of each pair element
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<__uint128_t> *MemData = *GetSrc<std::atomic<__uint128_t> **>(Data->SSAData, Op->Addr);
|
||||
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Expected);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Desired);
|
||||
|
||||
__uint128_t Expected = Src1;
|
||||
bool Result = MemData->compare_exchange_strong(Expected, Src2);
|
||||
memcpy(GDP, Result ? &Src1 : &Expected, 16);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", IROp->ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CAS) {
|
||||
auto Op = IROp->C<IR::IROp_CAS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint8_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint8_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint8_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint16_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint16_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint16_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint32_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint32_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint32_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
GD = AtomicCompareAndSwap(
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
|
||||
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
|
||||
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
|
||||
);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAdd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData += Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSub>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData -= Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAnd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData &= Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData |= Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
*MemData ^= Src;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->exchange(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_add(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_sub(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_and(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_or(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
|
||||
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint8_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
|
||||
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
|
||||
uint16_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
|
||||
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
uint64_t Previous = MemData->fetch_xor(Src);
|
||||
GD = Previous;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
using Type = uint8_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
using Type = uint16_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
using Type = uint32_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
using Type = uint64_t;
|
||||
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,160 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->SignalThread(Thread, FEXCore::Core::SignalEvent::Return);
|
||||
|
||||
LOGMAN_MSG_A_FMT("unreachable");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
SignalReturn(Data->State);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
Data->State->CurrentFrame->Pointers.Interpreter.CallbackReturn(Data->State, Data->StackEntry);
|
||||
}
|
||||
|
||||
DEF_OP(ExitFunction) {
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uintptr_t* ContextPtr = reinterpret_cast<uintptr_t*>(Data->State->CurrentFrame);
|
||||
|
||||
void *ContextData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->NewRIP);
|
||||
|
||||
memcpy(ContextData, Src, OpSize);
|
||||
|
||||
Data->BlockResults.Quit = true;
|
||||
}
|
||||
|
||||
DEF_OP(Jump) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
const uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TargetBlock);
|
||||
Data->BlockResults.Redo = true;
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
const uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
bool CompResult;
|
||||
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
if (Op->CompareSize == 4)
|
||||
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
|
||||
else
|
||||
CompResult = IsConditionTrue<uint64_t, int64_t, double>(Op->Cond.Val, Src1, Src2);
|
||||
|
||||
if (CompResult) {
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TrueBlock);
|
||||
}
|
||||
else {
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->FalseBlock);
|
||||
}
|
||||
Data->BlockResults.Redo = true;
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
|
||||
FEXCore::HLE::SyscallArguments Args;
|
||||
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
|
||||
if (Op->Header.Args[j].IsInvalid()) break;
|
||||
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
|
||||
}
|
||||
|
||||
uint64_t Res = FEXCore::Context::HandleSyscall(Data->State->CTX->SyscallHandler, Data->State->CurrentFrame, &Args);
|
||||
GD = Res;
|
||||
}
|
||||
|
||||
DEF_OP(InlineSyscall) {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
|
||||
FEXCore::HLE::SyscallArguments Args;
|
||||
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
|
||||
if (Op->Header.Args[j].IsInvalid()) break;
|
||||
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
|
||||
}
|
||||
|
||||
// We don't want the errno handling but I also don't want to write inline ASM atm
|
||||
uint64_t Res = syscall(
|
||||
Op->HostSyscallNumber,
|
||||
Args.Argument[0],
|
||||
Args.Argument[1],
|
||||
Args.Argument[2],
|
||||
Args.Argument[3],
|
||||
Args.Argument[4],
|
||||
Args.Argument[5],
|
||||
Args.Argument[6]
|
||||
);
|
||||
|
||||
if (Res == -1) {
|
||||
Res = -errno;
|
||||
}
|
||||
|
||||
GD = Res;
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
auto Op = IROp->C<IR::IROp_Thunk>();
|
||||
|
||||
auto thunkFn = Data->State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
thunkFn(*GetSrc<void**>(Data->SSAData, Op->ArgPtr));
|
||||
}
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
|
||||
auto CodePtr = Data->CurrentEntry + Op->Offset;
|
||||
if (memcmp((void*)CodePtr, &Op->CodeOriginalLow, Op->CodeLength) != 0) {
|
||||
GD = 1;
|
||||
} else {
|
||||
GD = 0;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
Data->State->CTX->ThreadRemoveCodeEntryFromJit(Data->State->CurrentFrame, Data->CurrentEntry);
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
const uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Function);
|
||||
const uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Leaf);
|
||||
|
||||
auto Results = Data->State->CTX->CPUID.RunFunction(Arg, Leaf);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,224 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->DestVector);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
Mask = ~0ULL;
|
||||
}
|
||||
Src2 = Src2 & Mask;
|
||||
Mask <<= Offset;
|
||||
Mask = ~Mask;
|
||||
__uint128_t Dst = Src1 & Mask;
|
||||
Dst |= Src2 << Offset;
|
||||
|
||||
memcpy(GDP, &Dst, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Src), Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
const float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
const float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
const double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
const double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
const double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Scalar);
|
||||
memcpy(GDP, &Dst, 8);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
const float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Scalar);
|
||||
memcpy(GDP, &Dst, 4);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- float
|
||||
// Only the lower elements from the source
|
||||
// This uses half the source elements
|
||||
uint8_t Elements = OpSize / 8;
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(double, float, Func, 0, 0)
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
// Little bit tricky here
|
||||
// Sometimes is used to convert from a 128bit vector register
|
||||
// in to a 64bit vector register with different sized elements
|
||||
// eg: %ssa5 i32v2 = Vector_FToF %ssa4 i128, #0x8
|
||||
uint8_t Elements = (OpSize << 1) / Op->SrcElementSize;
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
const auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
const auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
const auto Func_Trunc = [](auto a) { return std::trunc(a); };
|
||||
const auto Func_Host = [](auto a) { return std::rint(a); };
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Host)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Host)
|
||||
}
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,556 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace AES {
|
||||
static __uint128_t InvShiftRows(uint8_t *State) {
|
||||
uint8_t Shifted[16] = {
|
||||
State[0], State[13], State[10], State[7],
|
||||
State[4], State[1], State[14], State[11],
|
||||
State[8], State[5], State[2], State[15],
|
||||
State[12], State[9], State[6], State[3],
|
||||
};
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, Shifted, 16);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static __uint128_t InvSubBytes(uint8_t *State) {
|
||||
// 16x16 matrix table
|
||||
static const uint8_t InvSubstitutionTable[256] = {
|
||||
0x52, 0x09, 0x6a, 0xd5, 0x30, 0x36, 0xa5, 0x38, 0xbf, 0x40, 0xa3, 0x9e, 0x81, 0xf3, 0xd7, 0xfb,
|
||||
0x7c, 0xe3, 0x39, 0x82, 0x9b, 0x2f, 0xff, 0x87, 0x34, 0x8e, 0x43, 0x44, 0xc4, 0xde, 0xe9, 0xcb,
|
||||
0x54, 0x7b, 0x94, 0x32, 0xa6, 0xc2, 0x23, 0x3d, 0xee, 0x4c, 0x95, 0x0b, 0x42, 0xfa, 0xc3, 0x4e,
|
||||
0x08, 0x2e, 0xa1, 0x66, 0x28, 0xd9, 0x24, 0xb2, 0x76, 0x5b, 0xa2, 0x49, 0x6d, 0x8b, 0xd1, 0x25,
|
||||
0x72, 0xf8, 0xf6, 0x64, 0x86, 0x68, 0x98, 0x16, 0xd4, 0xa4, 0x5c, 0xcc, 0x5d, 0x65, 0xb6, 0x92,
|
||||
0x6c, 0x70, 0x48, 0x50, 0xfd, 0xed, 0xb9, 0xda, 0x5e, 0x15, 0x46, 0x57, 0xa7, 0x8d, 0x9d, 0x84,
|
||||
0x90, 0xd8, 0xab, 0x00, 0x8c, 0xbc, 0xd3, 0x0a, 0xf7, 0xe4, 0x58, 0x05, 0xb8, 0xb3, 0x45, 0x06,
|
||||
0xd0, 0x2c, 0x1e, 0x8f, 0xca, 0x3f, 0x0f, 0x02, 0xc1, 0xaf, 0xbd, 0x03, 0x01, 0x13, 0x8a, 0x6b,
|
||||
0x3a, 0x91, 0x11, 0x41, 0x4f, 0x67, 0xdc, 0xea, 0x97, 0xf2, 0xcf, 0xce, 0xf0, 0xb4, 0xe6, 0x73,
|
||||
0x96, 0xac, 0x74, 0x22, 0xe7, 0xad, 0x35, 0x85, 0xe2, 0xf9, 0x37, 0xe8, 0x1c, 0x75, 0xdf, 0x6e,
|
||||
0x47, 0xf1, 0x1a, 0x71, 0x1d, 0x29, 0xc5, 0x89, 0x6f, 0xb7, 0x62, 0x0e, 0xaa, 0x18, 0xbe, 0x1b,
|
||||
0xfc, 0x56, 0x3e, 0x4b, 0xc6, 0xd2, 0x79, 0x20, 0x9a, 0xdb, 0xc0, 0xfe, 0x78, 0xcd, 0x5a, 0xf4,
|
||||
0x1f, 0xdd, 0xa8, 0x33, 0x88, 0x07, 0xc7, 0x31, 0xb1, 0x12, 0x10, 0x59, 0x27, 0x80, 0xec, 0x5f,
|
||||
0x60, 0x51, 0x7f, 0xa9, 0x19, 0xb5, 0x4a, 0x0d, 0x2d, 0xe5, 0x7a, 0x9f, 0x93, 0xc9, 0x9c, 0xef,
|
||||
0xa0, 0xe0, 0x3b, 0x4d, 0xae, 0x2a, 0xf5, 0xb0, 0xc8, 0xeb, 0xbb, 0x3c, 0x83, 0x53, 0x99, 0x61,
|
||||
0x17, 0x2b, 0x04, 0x7e, 0xba, 0x77, 0xd6, 0x26, 0xe1, 0x69, 0x14, 0x63, 0x55, 0x21, 0x0c, 0x7d,
|
||||
};
|
||||
|
||||
// Uses a byte substitution table with a constant set of values
|
||||
// Needs to do a table look up
|
||||
uint8_t Substituted[16];
|
||||
for (size_t i = 0; i < 16; ++i) {
|
||||
Substituted[i] = InvSubstitutionTable[State[i]];
|
||||
}
|
||||
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, Substituted, 16);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static __uint128_t ShiftRows(uint8_t *State) {
|
||||
uint8_t Shifted[16] = {
|
||||
State[0], State[5], State[10], State[15],
|
||||
State[4], State[9], State[14], State[3],
|
||||
State[8], State[13], State[2], State[7],
|
||||
State[12], State[1], State[6], State[11],
|
||||
};
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, Shifted, 16);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static __uint128_t SubBytes(uint8_t *State, size_t Bytes) {
|
||||
// 16x16 matrix table
|
||||
static const uint8_t SubstitutionTable[256] = {
|
||||
0x63, 0x7c, 0x77, 0x7b, 0xf2, 0x6b, 0x6f, 0xc5, 0x30, 0x01, 0x67, 0x2b, 0xfe, 0xd7, 0xab, 0x76,
|
||||
0xca, 0x82, 0xc9, 0x7d, 0xfa, 0x59, 0x47, 0xf0, 0xad, 0xd4, 0xa2, 0xaf, 0x9c, 0xa4, 0x72, 0xc0,
|
||||
0xb7, 0xfd, 0x93, 0x26, 0x36, 0x3f, 0xf7, 0xcc, 0x34, 0xa5, 0xe5, 0xf1, 0x71, 0xd8, 0x31, 0x15,
|
||||
0x04, 0xc7, 0x23, 0xc3, 0x18, 0x96, 0x05, 0x9a, 0x07, 0x12, 0x80, 0xe2, 0xeb, 0x27, 0xb2, 0x75,
|
||||
0x09, 0x83, 0x2c, 0x1a, 0x1b, 0x6e, 0x5a, 0xa0, 0x52, 0x3b, 0xd6, 0xb3, 0x29, 0xe3, 0x2f, 0x84,
|
||||
0x53, 0xd1, 0x00, 0xed, 0x20, 0xfc, 0xb1, 0x5b, 0x6a, 0xcb, 0xbe, 0x39, 0x4a, 0x4c, 0x58, 0xcf,
|
||||
0xd0, 0xef, 0xaa, 0xfb, 0x43, 0x4d, 0x33, 0x85, 0x45, 0xf9, 0x02, 0x7f, 0x50, 0x3c, 0x9f, 0xa8,
|
||||
0x51, 0xa3, 0x40, 0x8f, 0x92, 0x9d, 0x38, 0xf5, 0xbc, 0xb6, 0xda, 0x21, 0x10, 0xff, 0xf3, 0xd2,
|
||||
0xcd, 0x0c, 0x13, 0xec, 0x5f, 0x97, 0x44, 0x17, 0xc4, 0xa7, 0x7e, 0x3d, 0x64, 0x5d, 0x19, 0x73,
|
||||
0x60, 0x81, 0x4f, 0xdc, 0x22, 0x2a, 0x90, 0x88, 0x46, 0xee, 0xb8, 0x14, 0xde, 0x5e, 0x0b, 0xdb,
|
||||
0xe0, 0x32, 0x3a, 0x0a, 0x49, 0x06, 0x24, 0x5c, 0xc2, 0xd3, 0xac, 0x62, 0x91, 0x95, 0xe4, 0x79,
|
||||
0xe7, 0xc8, 0x37, 0x6d, 0x8d, 0xd5, 0x4e, 0xa9, 0x6c, 0x56, 0xf4, 0xea, 0x65, 0x7a, 0xae, 0x08,
|
||||
0xba, 0x78, 0x25, 0x2e, 0x1c, 0xa6, 0xb4, 0xc6, 0xe8, 0xdd, 0x74, 0x1f, 0x4b, 0xbd, 0x8b, 0x8a,
|
||||
0x70, 0x3e, 0xb5, 0x66, 0x48, 0x03, 0xf6, 0x0e, 0x61, 0x35, 0x57, 0xb9, 0x86, 0xc1, 0x1d, 0x9e,
|
||||
0xe1, 0xf8, 0x98, 0x11, 0x69, 0xd9, 0x8e, 0x94, 0x9b, 0x1e, 0x87, 0xe9, 0xce, 0x55, 0x28, 0xdf,
|
||||
0x8c, 0xa1, 0x89, 0x0d, 0xbf, 0xe6, 0x42, 0x68, 0x41, 0x99, 0x2d, 0x0f, 0xb0, 0x54, 0xbb, 0x16,
|
||||
};
|
||||
// Uses a byte substitution table with a constant set of values
|
||||
// Needs to do a table look up
|
||||
uint8_t Substituted[16];
|
||||
Bytes = std::min(Bytes, (size_t)16);
|
||||
for (size_t i = 0; i < Bytes; ++i) {
|
||||
Substituted[i] = SubstitutionTable[State[i]];
|
||||
}
|
||||
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, Substituted, Bytes);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static uint8_t FFMul02(uint8_t in) {
|
||||
static const uint8_t FFMul02[256] = {
|
||||
0x00, 0x02, 0x04, 0x06, 0x08, 0x0a, 0x0c, 0x0e, 0x10, 0x12, 0x14, 0x16, 0x18, 0x1a, 0x1c, 0x1e,
|
||||
0x20, 0x22, 0x24, 0x26, 0x28, 0x2a, 0x2c, 0x2e, 0x30, 0x32, 0x34, 0x36, 0x38, 0x3a, 0x3c, 0x3e,
|
||||
0x40, 0x42, 0x44, 0x46, 0x48, 0x4a, 0x4c, 0x4e, 0x50, 0x52, 0x54, 0x56, 0x58, 0x5a, 0x5c, 0x5e,
|
||||
0x60, 0x62, 0x64, 0x66, 0x68, 0x6a, 0x6c, 0x6e, 0x70, 0x72, 0x74, 0x76, 0x78, 0x7a, 0x7c, 0x7e,
|
||||
0x80, 0x82, 0x84, 0x86, 0x88, 0x8a, 0x8c, 0x8e, 0x90, 0x92, 0x94, 0x96, 0x98, 0x9a, 0x9c, 0x9e,
|
||||
0xa0, 0xa2, 0xa4, 0xa6, 0xa8, 0xaa, 0xac, 0xae, 0xb0, 0xb2, 0xb4, 0xb6, 0xb8, 0xba, 0xbc, 0xbe,
|
||||
0xc0, 0xc2, 0xc4, 0xc6, 0xc8, 0xca, 0xcc, 0xce, 0xd0, 0xd2, 0xd4, 0xd6, 0xd8, 0xda, 0xdc, 0xde,
|
||||
0xe0, 0xe2, 0xe4, 0xe6, 0xe8, 0xea, 0xec, 0xee, 0xf0, 0xf2, 0xf4, 0xf6, 0xf8, 0xfa, 0xfc, 0xfe,
|
||||
0x1b, 0x19, 0x1f, 0x1d, 0x13, 0x11, 0x17, 0x15, 0x0b, 0x09, 0x0f, 0x0d, 0x03, 0x01, 0x07, 0x05,
|
||||
0x3b, 0x39, 0x3f, 0x3d, 0x33, 0x31, 0x37, 0x35, 0x2b, 0x29, 0x2f, 0x2d, 0x23, 0x21, 0x27, 0x25,
|
||||
0x5b, 0x59, 0x5f, 0x5d, 0x53, 0x51, 0x57, 0x55, 0x4b, 0x49, 0x4f, 0x4d, 0x43, 0x41, 0x47, 0x45,
|
||||
0x7b, 0x79, 0x7f, 0x7d, 0x73, 0x71, 0x77, 0x75, 0x6b, 0x69, 0x6f, 0x6d, 0x63, 0x61, 0x67, 0x65,
|
||||
0x9b, 0x99, 0x9f, 0x9d, 0x93, 0x91, 0x97, 0x95, 0x8b, 0x89, 0x8f, 0x8d, 0x83, 0x81, 0x87, 0x85,
|
||||
0xbb, 0xb9, 0xbf, 0xbd, 0xb3, 0xb1, 0xb7, 0xb5, 0xab, 0xa9, 0xaf, 0xad, 0xa3, 0xa1, 0xa7, 0xa5,
|
||||
0xdb, 0xd9, 0xdf, 0xdd, 0xd3, 0xd1, 0xd7, 0xd5, 0xcb, 0xc9, 0xcf, 0xcd, 0xc3, 0xc1, 0xc7, 0xc5,
|
||||
0xfb, 0xf9, 0xff, 0xfd, 0xf3, 0xf1, 0xf7, 0xf5, 0xeb, 0xe9, 0xef, 0xed, 0xe3, 0xe1, 0xe7, 0xe5,
|
||||
};
|
||||
return FFMul02[in];
|
||||
}
|
||||
|
||||
static uint8_t FFMul03(uint8_t in) {
|
||||
static const uint8_t FFMul03[256] = {
|
||||
0x00, 0x03, 0x06, 0x05, 0x0c, 0x0f, 0x0a, 0x09, 0x18, 0x1b, 0x1e, 0x1d, 0x14, 0x17, 0x12, 0x11,
|
||||
0x30, 0x33, 0x36, 0x35, 0x3c, 0x3f, 0x3a, 0x39, 0x28, 0x2b, 0x2e, 0x2d, 0x24, 0x27, 0x22, 0x21,
|
||||
0x60, 0x63, 0x66, 0x65, 0x6c, 0x6f, 0x6a, 0x69, 0x78, 0x7b, 0x7e, 0x7d, 0x74, 0x77, 0x72, 0x71,
|
||||
0x50, 0x53, 0x56, 0x55, 0x5c, 0x5f, 0x5a, 0x59, 0x48, 0x4b, 0x4e, 0x4d, 0x44, 0x47, 0x42, 0x41,
|
||||
0xc0, 0xc3, 0xc6, 0xc5, 0xcc, 0xcf, 0xca, 0xc9, 0xd8, 0xdb, 0xde, 0xdd, 0xd4, 0xd7, 0xd2, 0xd1,
|
||||
0xf0, 0xf3, 0xf6, 0xf5, 0xfc, 0xff, 0xfa, 0xf9, 0xe8, 0xeb, 0xee, 0xed, 0xe4, 0xe7, 0xe2, 0xe1,
|
||||
0xa0, 0xa3, 0xa6, 0xa5, 0xac, 0xaf, 0xaa, 0xa9, 0xb8, 0xbb, 0xbe, 0xbd, 0xb4, 0xb7, 0xb2, 0xb1,
|
||||
0x90, 0x93, 0x96, 0x95, 0x9c, 0x9f, 0x9a, 0x99, 0x88, 0x8b, 0x8e, 0x8d, 0x84, 0x87, 0x82, 0x81,
|
||||
0x9b, 0x98, 0x9d, 0x9e, 0x97, 0x94, 0x91, 0x92, 0x83, 0x80, 0x85, 0x86, 0x8f, 0x8c, 0x89, 0x8a,
|
||||
0xab, 0xa8, 0xad, 0xae, 0xa7, 0xa4, 0xa1, 0xa2, 0xb3, 0xb0, 0xb5, 0xb6, 0xbf, 0xbc, 0xb9, 0xba,
|
||||
0xfb, 0xf8, 0xfd, 0xfe, 0xf7, 0xf4, 0xf1, 0xf2, 0xe3, 0xe0, 0xe5, 0xe6, 0xef, 0xec, 0xe9, 0xea,
|
||||
0xcb, 0xc8, 0xcd, 0xce, 0xc7, 0xc4, 0xc1, 0xc2, 0xd3, 0xd0, 0xd5, 0xd6, 0xdf, 0xdc, 0xd9, 0xda,
|
||||
0x5b, 0x58, 0x5d, 0x5e, 0x57, 0x54, 0x51, 0x52, 0x43, 0x40, 0x45, 0x46, 0x4f, 0x4c, 0x49, 0x4a,
|
||||
0x6b, 0x68, 0x6d, 0x6e, 0x67, 0x64, 0x61, 0x62, 0x73, 0x70, 0x75, 0x76, 0x7f, 0x7c, 0x79, 0x7a,
|
||||
0x3b, 0x38, 0x3d, 0x3e, 0x37, 0x34, 0x31, 0x32, 0x23, 0x20, 0x25, 0x26, 0x2f, 0x2c, 0x29, 0x2a,
|
||||
0x0b, 0x08, 0x0d, 0x0e, 0x07, 0x04, 0x01, 0x02, 0x13, 0x10, 0x15, 0x16, 0x1f, 0x1c, 0x19, 0x1a,
|
||||
};
|
||||
return FFMul03[in];
|
||||
}
|
||||
|
||||
static __uint128_t MixColumns(uint8_t *State) {
|
||||
uint8_t In0[16] = {
|
||||
State[0], State[4], State[8], State[12],
|
||||
State[1], State[5], State[9], State[13],
|
||||
State[2], State[6], State[10], State[14],
|
||||
State[3], State[7], State[11], State[15],
|
||||
};
|
||||
|
||||
uint8_t Out0[4]{};
|
||||
uint8_t Out1[4]{};
|
||||
uint8_t Out2[4]{};
|
||||
uint8_t Out3[4]{};
|
||||
|
||||
for (size_t i = 0; i < 4; ++i) {
|
||||
Out0[i] = FFMul02(In0[0 + i]) ^ FFMul03(In0[4 + i]) ^ In0[8 + i] ^ In0[12 + i];
|
||||
Out1[i] = In0[0 + i] ^ FFMul02(In0[4 + i]) ^ FFMul03(In0[8 + i]) ^ In0[12 + i];
|
||||
Out2[i] = In0[0 + i] ^ In0[4 + i] ^ FFMul02(In0[8 + i]) ^ FFMul03(In0[12 + i]);
|
||||
Out3[i] = FFMul03(In0[0 + i]) ^ In0[4 + i] ^ In0[8 + i] ^ FFMul02(In0[12 + i]);
|
||||
}
|
||||
|
||||
uint8_t OutArray[16] = {
|
||||
Out0[0], Out1[0], Out2[0], Out3[0],
|
||||
Out0[1], Out1[1], Out2[1], Out3[1],
|
||||
Out0[2], Out1[2], Out2[2], Out3[2],
|
||||
Out0[3], Out1[3], Out2[3], Out3[3],
|
||||
};
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, OutArray, 16);
|
||||
return Res;
|
||||
}
|
||||
|
||||
static uint8_t FFMul09(uint8_t in) {
|
||||
static const uint8_t FFMul09[256] = {
|
||||
0x00, 0x09, 0x12, 0x1b, 0x24, 0x2d, 0x36, 0x3f, 0x48, 0x41, 0x5a, 0x53, 0x6c, 0x65, 0x7e, 0x77,
|
||||
0x90, 0x99, 0x82, 0x8b, 0xb4, 0xbd, 0xa6, 0xaf, 0xd8, 0xd1, 0xca, 0xc3, 0xfc, 0xf5, 0xee, 0xe7,
|
||||
0x3b, 0x32, 0x29, 0x20, 0x1f, 0x16, 0x0d, 0x04, 0x73, 0x7a, 0x61, 0x68, 0x57, 0x5e, 0x45, 0x4c,
|
||||
0xab, 0xa2, 0xb9, 0xb0, 0x8f, 0x86, 0x9d, 0x94, 0xe3, 0xea, 0xf1, 0xf8, 0xc7, 0xce, 0xd5, 0xdc,
|
||||
0x76, 0x7f, 0x64, 0x6d, 0x52, 0x5b, 0x40, 0x49, 0x3e, 0x37, 0x2c, 0x25, 0x1a, 0x13, 0x08, 0x01,
|
||||
0xe6, 0xef, 0xf4, 0xfd, 0xc2, 0xcb, 0xd0, 0xd9, 0xae, 0xa7, 0xbc, 0xb5, 0x8a, 0x83, 0x98, 0x91,
|
||||
0x4d, 0x44, 0x5f, 0x56, 0x69, 0x60, 0x7b, 0x72, 0x05, 0x0c, 0x17, 0x1e, 0x21, 0x28, 0x33, 0x3a,
|
||||
0xdd, 0xd4, 0xcf, 0xc6, 0xf9, 0xf0, 0xeb, 0xe2, 0x95, 0x9c, 0x87, 0x8e, 0xb1, 0xb8, 0xa3, 0xaa,
|
||||
0xec, 0xe5, 0xfe, 0xf7, 0xc8, 0xc1, 0xda, 0xd3, 0xa4, 0xad, 0xb6, 0xbf, 0x80, 0x89, 0x92, 0x9b,
|
||||
0x7c, 0x75, 0x6e, 0x67, 0x58, 0x51, 0x4a, 0x43, 0x34, 0x3d, 0x26, 0x2f, 0x10, 0x19, 0x02, 0x0b,
|
||||
0xd7, 0xde, 0xc5, 0xcc, 0xf3, 0xfa, 0xe1, 0xe8, 0x9f, 0x96, 0x8d, 0x84, 0xbb, 0xb2, 0xa9, 0xa0,
|
||||
0x47, 0x4e, 0x55, 0x5c, 0x63, 0x6a, 0x71, 0x78, 0x0f, 0x06, 0x1d, 0x14, 0x2b, 0x22, 0x39, 0x30,
|
||||
0x9a, 0x93, 0x88, 0x81, 0xbe, 0xb7, 0xac, 0xa5, 0xd2, 0xdb, 0xc0, 0xc9, 0xf6, 0xff, 0xe4, 0xed,
|
||||
0x0a, 0x03, 0x18, 0x11, 0x2e, 0x27, 0x3c, 0x35, 0x42, 0x4b, 0x50, 0x59, 0x66, 0x6f, 0x74, 0x7d,
|
||||
0xa1, 0xa8, 0xb3, 0xba, 0x85, 0x8c, 0x97, 0x9e, 0xe9, 0xe0, 0xfb, 0xf2, 0xcd, 0xc4, 0xdf, 0xd6,
|
||||
0x31, 0x38, 0x23, 0x2a, 0x15, 0x1c, 0x07, 0x0e, 0x79, 0x70, 0x6b, 0x62, 0x5d, 0x54, 0x4f, 0x46,
|
||||
};
|
||||
return FFMul09[in];
|
||||
}
|
||||
|
||||
static uint8_t FFMul0B(uint8_t in) {
|
||||
static const uint8_t FFMul0B[256] = {
|
||||
0x00, 0x0b, 0x16, 0x1d, 0x2c, 0x27, 0x3a, 0x31, 0x58, 0x53, 0x4e, 0x45, 0x74, 0x7f, 0x62, 0x69,
|
||||
0xb0, 0xbb, 0xa6, 0xad, 0x9c, 0x97, 0x8a, 0x81, 0xe8, 0xe3, 0xfe, 0xf5, 0xc4, 0xcf, 0xd2, 0xd9,
|
||||
0x7b, 0x70, 0x6d, 0x66, 0x57, 0x5c, 0x41, 0x4a, 0x23, 0x28, 0x35, 0x3e, 0x0f, 0x04, 0x19, 0x12,
|
||||
0xcb, 0xc0, 0xdd, 0xd6, 0xe7, 0xec, 0xf1, 0xfa, 0x93, 0x98, 0x85, 0x8e, 0xbf, 0xb4, 0xa9, 0xa2,
|
||||
0xf6, 0xfd, 0xe0, 0xeb, 0xda, 0xd1, 0xcc, 0xc7, 0xae, 0xa5, 0xb8, 0xb3, 0x82, 0x89, 0x94, 0x9f,
|
||||
0x46, 0x4d, 0x50, 0x5b, 0x6a, 0x61, 0x7c, 0x77, 0x1e, 0x15, 0x08, 0x03, 0x32, 0x39, 0x24, 0x2f,
|
||||
0x8d, 0x86, 0x9b, 0x90, 0xa1, 0xaa, 0xb7, 0xbc, 0xd5, 0xde, 0xc3, 0xc8, 0xf9, 0xf2, 0xef, 0xe4,
|
||||
0x3d, 0x36, 0x2b, 0x20, 0x11, 0x1a, 0x07, 0x0c, 0x65, 0x6e, 0x73, 0x78, 0x49, 0x42, 0x5f, 0x54,
|
||||
0xf7, 0xfc, 0xe1, 0xea, 0xdb, 0xd0, 0xcd, 0xc6, 0xaf, 0xa4, 0xb9, 0xb2, 0x83, 0x88, 0x95, 0x9e,
|
||||
0x47, 0x4c, 0x51, 0x5a, 0x6b, 0x60, 0x7d, 0x76, 0x1f, 0x14, 0x09, 0x02, 0x33, 0x38, 0x25, 0x2e,
|
||||
0x8c, 0x87, 0x9a, 0x91, 0xa0, 0xab, 0xb6, 0xbd, 0xd4, 0xdf, 0xc2, 0xc9, 0xf8, 0xf3, 0xee, 0xe5,
|
||||
0x3c, 0x37, 0x2a, 0x21, 0x10, 0x1b, 0x06, 0x0d, 0x64, 0x6f, 0x72, 0x79, 0x48, 0x43, 0x5e, 0x55,
|
||||
0x01, 0x0a, 0x17, 0x1c, 0x2d, 0x26, 0x3b, 0x30, 0x59, 0x52, 0x4f, 0x44, 0x75, 0x7e, 0x63, 0x68,
|
||||
0xb1, 0xba, 0xa7, 0xac, 0x9d, 0x96, 0x8b, 0x80, 0xe9, 0xe2, 0xff, 0xf4, 0xc5, 0xce, 0xd3, 0xd8,
|
||||
0x7a, 0x71, 0x6c, 0x67, 0x56, 0x5d, 0x40, 0x4b, 0x22, 0x29, 0x34, 0x3f, 0x0e, 0x05, 0x18, 0x13,
|
||||
0xca, 0xc1, 0xdc, 0xd7, 0xe6, 0xed, 0xf0, 0xfb, 0x92, 0x99, 0x84, 0x8f, 0xbe, 0xb5, 0xa8, 0xa3,
|
||||
};
|
||||
return FFMul0B[in];
|
||||
}
|
||||
|
||||
static uint8_t FFMul0D(uint8_t in) {
|
||||
static const uint8_t FFMul0D[256] = {
|
||||
0x00, 0x0d, 0x1a, 0x17, 0x34, 0x39, 0x2e, 0x23, 0x68, 0x65, 0x72, 0x7f, 0x5c, 0x51, 0x46, 0x4b,
|
||||
0xd0, 0xdd, 0xca, 0xc7, 0xe4, 0xe9, 0xfe, 0xf3, 0xb8, 0xb5, 0xa2, 0xaf, 0x8c, 0x81, 0x96, 0x9b,
|
||||
0xbb, 0xb6, 0xa1, 0xac, 0x8f, 0x82, 0x95, 0x98, 0xd3, 0xde, 0xc9, 0xc4, 0xe7, 0xea, 0xfd, 0xf0,
|
||||
0x6b, 0x66, 0x71, 0x7c, 0x5f, 0x52, 0x45, 0x48, 0x03, 0x0e, 0x19, 0x14, 0x37, 0x3a, 0x2d, 0x20,
|
||||
0x6d, 0x60, 0x77, 0x7a, 0x59, 0x54, 0x43, 0x4e, 0x05, 0x08, 0x1f, 0x12, 0x31, 0x3c, 0x2b, 0x26,
|
||||
0xbd, 0xb0, 0xa7, 0xaa, 0x89, 0x84, 0x93, 0x9e, 0xd5, 0xd8, 0xcf, 0xc2, 0xe1, 0xec, 0xfb, 0xf6,
|
||||
0xd6, 0xdb, 0xcc, 0xc1, 0xe2, 0xef, 0xf8, 0xf5, 0xbe, 0xb3, 0xa4, 0xa9, 0x8a, 0x87, 0x90, 0x9d,
|
||||
0x06, 0x0b, 0x1c, 0x11, 0x32, 0x3f, 0x28, 0x25, 0x6e, 0x63, 0x74, 0x79, 0x5a, 0x57, 0x40, 0x4d,
|
||||
0xda, 0xd7, 0xc0, 0xcd, 0xee, 0xe3, 0xf4, 0xf9, 0xb2, 0xbf, 0xa8, 0xa5, 0x86, 0x8b, 0x9c, 0x91,
|
||||
0x0a, 0x07, 0x10, 0x1d, 0x3e, 0x33, 0x24, 0x29, 0x62, 0x6f, 0x78, 0x75, 0x56, 0x5b, 0x4c, 0x41,
|
||||
0x61, 0x6c, 0x7b, 0x76, 0x55, 0x58, 0x4f, 0x42, 0x09, 0x04, 0x13, 0x1e, 0x3d, 0x30, 0x27, 0x2a,
|
||||
0xb1, 0xbc, 0xab, 0xa6, 0x85, 0x88, 0x9f, 0x92, 0xd9, 0xd4, 0xc3, 0xce, 0xed, 0xe0, 0xf7, 0xfa,
|
||||
0xb7, 0xba, 0xad, 0xa0, 0x83, 0x8e, 0x99, 0x94, 0xdf, 0xd2, 0xc5, 0xc8, 0xeb, 0xe6, 0xf1, 0xfc,
|
||||
0x67, 0x6a, 0x7d, 0x70, 0x53, 0x5e, 0x49, 0x44, 0x0f, 0x02, 0x15, 0x18, 0x3b, 0x36, 0x21, 0x2c,
|
||||
0x0c, 0x01, 0x16, 0x1b, 0x38, 0x35, 0x22, 0x2f, 0x64, 0x69, 0x7e, 0x73, 0x50, 0x5d, 0x4a, 0x47,
|
||||
0xdc, 0xd1, 0xc6, 0xcb, 0xe8, 0xe5, 0xf2, 0xff, 0xb4, 0xb9, 0xae, 0xa3, 0x80, 0x8d, 0x9a, 0x97,
|
||||
};
|
||||
|
||||
return FFMul0D[in];
|
||||
}
|
||||
|
||||
static uint8_t FFMul0E(uint8_t in) {
|
||||
static const uint8_t FFMul0E[256] = {
|
||||
0x00, 0x0e, 0x1c, 0x12, 0x38, 0x36, 0x24, 0x2a, 0x70, 0x7e, 0x6c, 0x62, 0x48, 0x46, 0x54, 0x5a,
|
||||
0xe0, 0xee, 0xfc, 0xf2, 0xd8, 0xd6, 0xc4, 0xca, 0x90, 0x9e, 0x8c, 0x82, 0xa8, 0xa6, 0xb4, 0xba,
|
||||
0xdb, 0xd5, 0xc7, 0xc9, 0xe3, 0xed, 0xff, 0xf1, 0xab, 0xa5, 0xb7, 0xb9, 0x93, 0x9d, 0x8f, 0x81,
|
||||
0x3b, 0x35, 0x27, 0x29, 0x03, 0x0d, 0x1f, 0x11, 0x4b, 0x45, 0x57, 0x59, 0x73, 0x7d, 0x6f, 0x61,
|
||||
0xad, 0xa3, 0xb1, 0xbf, 0x95, 0x9b, 0x89, 0x87, 0xdd, 0xd3, 0xc1, 0xcf, 0xe5, 0xeb, 0xf9, 0xf7,
|
||||
0x4d, 0x43, 0x51, 0x5f, 0x75, 0x7b, 0x69, 0x67, 0x3d, 0x33, 0x21, 0x2f, 0x05, 0x0b, 0x19, 0x17,
|
||||
0x76, 0x78, 0x6a, 0x64, 0x4e, 0x40, 0x52, 0x5c, 0x06, 0x08, 0x1a, 0x14, 0x3e, 0x30, 0x22, 0x2c,
|
||||
0x96, 0x98, 0x8a, 0x84, 0xae, 0xa0, 0xb2, 0xbc, 0xe6, 0xe8, 0xfa, 0xf4, 0xde, 0xd0, 0xc2, 0xcc,
|
||||
0x41, 0x4f, 0x5d, 0x53, 0x79, 0x77, 0x65, 0x6b, 0x31, 0x3f, 0x2d, 0x23, 0x09, 0x07, 0x15, 0x1b,
|
||||
0xa1, 0xaf, 0xbd, 0xb3, 0x99, 0x97, 0x85, 0x8b, 0xd1, 0xdf, 0xcd, 0xc3, 0xe9, 0xe7, 0xf5, 0xfb,
|
||||
0x9a, 0x94, 0x86, 0x88, 0xa2, 0xac, 0xbe, 0xb0, 0xea, 0xe4, 0xf6, 0xf8, 0xd2, 0xdc, 0xce, 0xc0,
|
||||
0x7a, 0x74, 0x66, 0x68, 0x42, 0x4c, 0x5e, 0x50, 0x0a, 0x04, 0x16, 0x18, 0x32, 0x3c, 0x2e, 0x20,
|
||||
0xec, 0xe2, 0xf0, 0xfe, 0xd4, 0xda, 0xc8, 0xc6, 0x9c, 0x92, 0x80, 0x8e, 0xa4, 0xaa, 0xb8, 0xb6,
|
||||
0x0c, 0x02, 0x10, 0x1e, 0x34, 0x3a, 0x28, 0x26, 0x7c, 0x72, 0x60, 0x6e, 0x44, 0x4a, 0x58, 0x56,
|
||||
0x37, 0x39, 0x2b, 0x25, 0x0f, 0x01, 0x13, 0x1d, 0x47, 0x49, 0x5b, 0x55, 0x7f, 0x71, 0x63, 0x6d,
|
||||
0xd7, 0xd9, 0xcb, 0xc5, 0xef, 0xe1, 0xf3, 0xfd, 0xa7, 0xa9, 0xbb, 0xb5, 0x9f, 0x91, 0x83, 0x8d,
|
||||
};
|
||||
|
||||
return FFMul0E[in];
|
||||
}
|
||||
|
||||
static __uint128_t InvMixColumns(uint8_t *State) {
|
||||
uint8_t In0[16] = {
|
||||
State[0], State[4], State[8], State[12],
|
||||
State[1], State[5], State[9], State[13],
|
||||
State[2], State[6], State[10], State[14],
|
||||
State[3], State[7], State[11], State[15],
|
||||
};
|
||||
|
||||
uint8_t Out0[4]{};
|
||||
uint8_t Out1[4]{};
|
||||
uint8_t Out2[4]{};
|
||||
uint8_t Out3[4]{};
|
||||
|
||||
for (size_t i = 0; i < 4; ++i) {
|
||||
Out0[i] = FFMul0E(In0[0 + i]) ^ FFMul0B(In0[4 + i]) ^ FFMul0D(In0[8 + i]) ^ FFMul09(In0[12 + i]);
|
||||
Out1[i] = FFMul09(In0[0 + i]) ^ FFMul0E(In0[4 + i]) ^ FFMul0B(In0[8 + i]) ^ FFMul0D(In0[12 + i]);
|
||||
Out2[i] = FFMul0D(In0[0 + i]) ^ FFMul09(In0[4 + i]) ^ FFMul0E(In0[8 + i]) ^ FFMul0B(In0[12 + i]);
|
||||
Out3[i] = FFMul0B(In0[0 + i]) ^ FFMul0D(In0[4 + i]) ^ FFMul09(In0[8 + i]) ^ FFMul0E(In0[12 + i]);
|
||||
}
|
||||
|
||||
uint8_t OutArray[16] = {
|
||||
Out0[0], Out1[0], Out2[0], Out3[0],
|
||||
Out0[1], Out1[1], Out2[1], Out3[1],
|
||||
Out0[2], Out1[2], Out2[2], Out3[2],
|
||||
Out0[3], Out1[3], Out2[3], Out3[3],
|
||||
};
|
||||
__uint128_t Res{};
|
||||
memcpy(&Res, OutArray, 16);
|
||||
return Res;
|
||||
}
|
||||
}
|
||||
|
||||
namespace CRC32 {
|
||||
// CRC32 per byte lookup table.
|
||||
constexpr std::array<uint32_t, 256> CRC32CTable = []() consteval {
|
||||
std::array<uint32_t, 256> Table{};
|
||||
|
||||
// Clang 11.x doesn't support bitreverse as a consteval
|
||||
// constexpr uint32_t Polynomial = 0x1EDC6F41;
|
||||
constexpr uint32_t PolynomialRev = 0x82F63B78; //__builtin_bitreverse32(Polynomial);
|
||||
|
||||
for (size_t Char = 0; Char < std::size(Table); ++Char) {
|
||||
uint32_t CurrentChar = Char;
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
if (CurrentChar & 1) {
|
||||
CurrentChar = (CurrentChar >> 1) ^ PolynomialRev;
|
||||
}
|
||||
else {
|
||||
CurrentChar >>= 1;
|
||||
}
|
||||
}
|
||||
Table[Char] = CurrentChar;
|
||||
}
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
uint32_t crc32cb(uint32_t Accumulator, uint8_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ data] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32ch(uint32_t Accumulator, uint16_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32cw(uint32_t Accumulator, uint32_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32cx(uint32_t Accumulator, uint64_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 32) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 40) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 48) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 56) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
|
||||
// Pseudo-code
|
||||
// Dst = InvMixColumns(STATE)
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Src1));
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
// RoundKey = Src2
|
||||
// STATE = ShiftRows(STATE)
|
||||
// STATE = SubBytes(STATE)
|
||||
// STATE = MixColumns(STATE)
|
||||
// Dst = STATE XOR RoundKey
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
|
||||
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
|
||||
Tmp = AES::MixColumns(reinterpret_cast<uint8_t*>(&Tmp));
|
||||
Tmp = Tmp ^ Src2;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
// RoundKey = Src2
|
||||
// STATE = ShiftRows(STATE)
|
||||
// STATE = SubBytes(STATE)
|
||||
// Dst = STATE XOR RoundKey
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
|
||||
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
|
||||
Tmp = Tmp ^ Src2;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
// RoundKey = Src2
|
||||
// STATE = InvShiftRows(STATE)
|
||||
// STATE = InvSubBytes(STATE)
|
||||
// STATE = InvMixColumns(STATE)
|
||||
// Dst = STATE XOR RoundKey
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
|
||||
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
|
||||
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Tmp));
|
||||
Tmp = Tmp ^ Src2;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
// RoundKey = Src2
|
||||
// STATE = InvShiftRows(STATE)
|
||||
// STATE = InvSubBytes(STATE)
|
||||
// Dst = STATE XOR RoundKey
|
||||
__uint128_t Tmp{};
|
||||
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
|
||||
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
|
||||
Tmp = Tmp ^ Src2;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
// Pseudo-code
|
||||
// X3 = Src1[127:96]
|
||||
// X2 = Src1[95:64]
|
||||
// X1 = Src1[63:32]
|
||||
// X0 = Src1[31:30]
|
||||
// RCON = (Zext)rcon
|
||||
// Dest[31:0] = SubWord(X1)
|
||||
// Dest[63:32] = RotWord(SubWord(X1)) XOR RCON
|
||||
// Dest[95:64] = SubWord(X3)
|
||||
// Dest[127:96] = RotWord(SubWord(X3)) XOR RCON
|
||||
__uint128_t Tmp{};
|
||||
uint32_t X1{};
|
||||
uint32_t X3{};
|
||||
memcpy(&X1, &Src1[4], 4);
|
||||
memcpy(&X3, &Src1[12], 4);
|
||||
uint32_t SubWord_X1 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X1), 4);
|
||||
uint32_t SubWord_X3 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X3), 4);
|
||||
|
||||
auto Ror = [] (auto In, auto R) {
|
||||
auto RotateMask = sizeof(In) * 8 - 1;
|
||||
R &= RotateMask;
|
||||
return (In >> R) | (In << (sizeof(In) * 8 - R));
|
||||
};
|
||||
|
||||
uint32_t Rot_X1 = Ror(SubWord_X1, 8);
|
||||
uint32_t Rot_X3 = Ror(SubWord_X3, 8);
|
||||
|
||||
Tmp = Rot_X3 ^ Op->RCON;
|
||||
Tmp <<= 32;
|
||||
Tmp |= SubWord_X3;
|
||||
Tmp <<= 32;
|
||||
Tmp |= Rot_X1 ^ Op->RCON;
|
||||
Tmp <<= 32;
|
||||
Tmp |= SubWord_X1;
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
uint32_t Src1 = *GetSrc<uint32_t*>(Data->SSAData, Op->Src1);
|
||||
uint8_t *Src2 = GetSrc<uint8_t*>(Data->SSAData, Op->Src2);
|
||||
uint32_t Tmp{};
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
Tmp = CRC32::crc32cb(Src1, *(uint8_t*)Src2);
|
||||
break;
|
||||
case 2:
|
||||
Tmp = CRC32::crc32ch(Src1, *(uint16_t*)Src2);
|
||||
break;
|
||||
case 4:
|
||||
Tmp = CRC32::crc32cw(Src1, *(uint32_t*)Src2);
|
||||
break;
|
||||
case 8:
|
||||
Tmp = CRC32::crc32cx(Src1, *(uint64_t*)Src2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown CRC32C size: {}", Op->SrcSize);
|
||||
break;
|
||||
|
||||
}
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
const auto Selector = Op->Selector;
|
||||
auto* Dst = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
auto* Src1 = GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
auto* Src2 = GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
const uint64_t TMP1 = (Selector & 0x01) == 0 ? Src1[0] : Src1[1];
|
||||
const uint64_t TMP2 = (Selector & 0x10) == 0 ? Src2[0] : Src2[1];
|
||||
|
||||
const auto make_lo = [](uint64_t lhs, uint64_t rhs) {
|
||||
uint64_t result = 0;
|
||||
|
||||
for (size_t i = 0; i < 64; i++) {
|
||||
if ((lhs & (1ULL << i)) != 0) {
|
||||
result ^= rhs << i;
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
};
|
||||
const auto make_hi = [](uint64_t lhs, uint64_t rhs) {
|
||||
uint64_t result = 0;
|
||||
|
||||
for (size_t i = 1; i < 64; i++) {
|
||||
if ((lhs & (1ULL << i)) != 0) {
|
||||
result ^= rhs >> (64 - i);
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
};
|
||||
|
||||
Dst[0] = make_lo(TMP1, TMP2);
|
||||
Dst[1] = make_hi(TMP1, TMP2);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,423 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include "F80Ops.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(F80LOADFCW) {
|
||||
FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle(*GetSrc<uint16_t*>(Data->SSAData, IROp->Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(F80ADD) {
|
||||
auto Op = IROp->C<IR::IROp_F80Add>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FADD(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SUB) {
|
||||
auto Op = IROp->C<IR::IROp_F80Sub>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSUB(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80MUL) {
|
||||
auto Op = IROp->C<IR::IROp_F80Mul>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FMUL(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80DIV) {
|
||||
auto Op = IROp->C<IR::IROp_F80Div>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FDIV(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F80FYL2X>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FYL2X(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80ATAN>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FATAN(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM1>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM1(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F80SCALE>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSCALE(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CVT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVT>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
float Tmp = Src;
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
double Tmp = Src;
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80CVTINT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
int16_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2)(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4)(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
int64_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8)(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80CVTTO) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
float Src = *GetSrc<float *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
double Src = *GetSrc<double *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80CVTTOINT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F80ROUND) {
|
||||
auto Op = IROp->C<IR::IROp_F80Round>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FRNDINT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80F2XM1>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::F2XM1(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80TAN>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FTAN(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SQRT) {
|
||||
auto Op = IROp->C<IR::IROp_F80SQRT>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSQRT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F80SIN>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSIN(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80COS) {
|
||||
auto Op = IROp->C<IR::IROp_F80COS>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FCOS(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_EXP) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_EXP(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_SIG) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_SIG(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CMP) {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
uint32_t ResultFlags{};
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
bool eq, lt, nan;
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
lt) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
}
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
|
||||
nan) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
|
||||
}
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ) &&
|
||||
eq) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
|
||||
}
|
||||
|
||||
GD = ResultFlags;
|
||||
}
|
||||
|
||||
DEF_OP(F80BCDLOAD) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
|
||||
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->X80Src);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
// Only 0-9 is supported, A-F results in undefined data
|
||||
// | 4 bit | 4 bit |
|
||||
// | 10s place | 1s place |
|
||||
// EG 0x48 = 48
|
||||
// EG 0x4847 = 4847
|
||||
// This gives us an 18digit value encoded in BCD
|
||||
// The last byte lets us know if it negative or not
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
uint8_t Digit = Src1[8 - i];
|
||||
// First shift our last value over
|
||||
BCD *= 100;
|
||||
|
||||
// Add the tens place digit
|
||||
BCD += (Digit >> 4) * 10;
|
||||
|
||||
// Add the ones place digit
|
||||
BCD += Digit & 0xF;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
bool Negative = Src1[9] & 0x80;
|
||||
X80SoftFloat Tmp;
|
||||
|
||||
Tmp = BCD;
|
||||
Tmp.Sign = Negative;
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80BCDSTORE) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDStore>();
|
||||
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src));
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Sign = 0;
|
||||
|
||||
uint64_t Tmp = Src1;
|
||||
uint8_t BCD[10]{};
|
||||
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
if (Tmp == 0) {
|
||||
// Nothing left? Just leave
|
||||
break;
|
||||
}
|
||||
// Extract the lower 100 values
|
||||
uint8_t Digit = Tmp % 100;
|
||||
|
||||
// Now divide it for the next iteration
|
||||
Tmp /= 100;
|
||||
|
||||
uint8_t UpperNibble = Digit / 10;
|
||||
uint8_t LowerNibble = Digit % 10;
|
||||
|
||||
// Now store the BCD
|
||||
BCD[i] = (UpperNibble << 4) | LowerNibble;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
BCD[9] = Negative ? 0x80 : 0;
|
||||
|
||||
memcpy(GDP, BCD, 10);
|
||||
}
|
||||
|
||||
DEF_OP(F64SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F64SIN>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = sin(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64COS) {
|
||||
auto Op = IROp->C<IR::IROp_F64COS>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = cos(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64TAN>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = tan(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64F2XM1>();
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = exp2(Src) - 1.0;
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64ATAN>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = atan2(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = fmod(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM1>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = remainder(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F64FYL2X>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = Src2 * log2(Src1);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F64SCALE>();
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double trunc = (double)(int64_t)(Src2); //truncate
|
||||
const double Tmp = Src1 * exp2(trunc);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,399 +0,0 @@
|
||||
#pragma once
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/SoftFloat-3e/softfloat.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
template<IR::IROps Op>
|
||||
struct OpHandlers {
|
||||
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
static X80SoftFloat handle4(float src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static X80SoftFloat handle8(double src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CMP> {
|
||||
template<uint32_t Flags>
|
||||
static uint64_t handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
bool eq, lt, nan;
|
||||
uint64_t ResultFlags = 0;
|
||||
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
lt) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
|
||||
nan) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_EQ) &&
|
||||
eq) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
|
||||
}
|
||||
return ResultFlags;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVT> {
|
||||
static float handle4(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static double handle8(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
static int16_t handle2(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static int32_t handle4(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static int64_t handle8(X80SoftFloat src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static int16_t handle2t(X80SoftFloat src) {
|
||||
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
|
||||
if (rv > INT16_MAX) {
|
||||
return INT16_MAX;
|
||||
} else if (rv < INT16_MIN) {
|
||||
return INT16_MIN;
|
||||
} else {
|
||||
return rv;
|
||||
}
|
||||
}
|
||||
|
||||
static int32_t handle4t(X80SoftFloat src) {
|
||||
return extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
}
|
||||
|
||||
static int64_t handle8t(X80SoftFloat src) {
|
||||
return extF80_to_i64(src, softfloat_round_minMag, false);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
static X80SoftFloat handle2(int16_t src) {
|
||||
return src;
|
||||
}
|
||||
|
||||
static X80SoftFloat handle4(int32_t src) {
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FRNDINT(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::F2XM1(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FTAN(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SQRT> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FSQRT(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FSIN(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FCOS(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FXTRACT_EXP(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
return X80SoftFloat::FXTRACT_SIG(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ADD> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FADD(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SUB> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FSUB(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80MUL> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FMUL(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80DIV> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FDIV(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FYL2X(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FATAN(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FREM1(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FREM(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
return X80SoftFloat::FSCALE(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
static double handle(double src) {
|
||||
return sin(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
static double handle(double src) {
|
||||
return cos(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
static double handle(double src) {
|
||||
return tan(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
static double handle(double src) {
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
static double handle(double src1, double src2) {
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
static double handle(double src1, double src2) {
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
static double handle(double src1, double src2) {
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
static double handle(double src1, double src2) {
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(double src1, double src2) {
|
||||
double trunc = (double)(int64_t)(src2); //truncate
|
||||
return src1 * exp2(trunc);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
Src1 = X80SoftFloat::FRNDINT(Src1);
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Sign = 0;
|
||||
|
||||
uint64_t Tmp = Src1;
|
||||
X80SoftFloat Rv;
|
||||
uint8_t *BCD = reinterpret_cast<uint8_t*>(&Rv);
|
||||
memset(BCD, 0, 10);
|
||||
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
if (Tmp == 0) {
|
||||
// Nothing left? Just leave
|
||||
break;
|
||||
}
|
||||
// Extract the lower 100 values
|
||||
uint8_t Digit = Tmp % 100;
|
||||
|
||||
// Now divide it for the next iteration
|
||||
Tmp /= 100;
|
||||
|
||||
uint8_t UpperNibble = Digit / 10;
|
||||
uint8_t LowerNibble = Digit % 10;
|
||||
|
||||
// Now store the BCD
|
||||
BCD[i] = (UpperNibble << 4) | LowerNibble;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
BCD[9] = Negative ? 0x80 : 0;
|
||||
|
||||
return Rv;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src) {
|
||||
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
// Only 0-9 is supported, A-F results in undefined data
|
||||
// | 4 bit | 4 bit |
|
||||
// | 10s place | 1s place |
|
||||
// EG 0x48 = 48
|
||||
// EG 0x4847 = 4847
|
||||
// This gives us an 18digit value encoded in BCD
|
||||
// The last byte lets us know if it negative or not
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
uint8_t Digit = Src1[8 - i];
|
||||
// First shift our last value over
|
||||
BCD *= 100;
|
||||
|
||||
// Add the tens place digit
|
||||
BCD += (Digit >> 4) * 10;
|
||||
|
||||
// Add the ones place digit
|
||||
BCD += Digit & 0xF;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
bool Negative = Src1[9] & 0x80;
|
||||
X80SoftFloat Tmp;
|
||||
|
||||
Tmp = BCD;
|
||||
Tmp.Sign = Negative;
|
||||
return Tmp;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80LOADFCW> {
|
||||
static void handle(uint16_t NewFCW) {
|
||||
|
||||
auto PC = (NewFCW >> 8) & 3;
|
||||
switch(PC) {
|
||||
case 0: extF80_roundingPrecision = 32; break;
|
||||
case 2: extF80_roundingPrecision = 64; break;
|
||||
case 3: extF80_roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
}
|
||||
|
||||
auto RC = (NewFCW >> 10) & 3;
|
||||
switch(RC) {
|
||||
case 0:
|
||||
softfloat_roundingMode = softfloat_round_near_even;
|
||||
break;
|
||||
case 1:
|
||||
softfloat_roundingMode = softfloat_round_min;
|
||||
break;
|
||||
case 2:
|
||||
softfloat_roundingMode = softfloat_round_max;
|
||||
break;
|
||||
case 3:
|
||||
softfloat_roundingMode = softfloat_round_minMag;
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
}
|
||||
@@ -1,21 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Value) >> Op->Flag) & 1;
|
||||
}
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,5 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
@@ -8,7 +9,6 @@
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Dispatcher;
|
||||
class X86DispatchGenerator;
|
||||
class Arm64DispatchGenerator;
|
||||
|
||||
@@ -21,35 +21,32 @@ using DestMapType = std::vector<uint32_t>;
|
||||
|
||||
class InterpreterCore final : public CPUBackend {
|
||||
public:
|
||||
explicit InterpreterCore(Dispatcher *Dispatch,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
explicit InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
~InterpreterCore() override;
|
||||
std::string GetName() override { return "Interpreter"; }
|
||||
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearCache() override;
|
||||
bool HandleSIGBUS(int Signal, void *info, void *ucontext);
|
||||
|
||||
private:
|
||||
size_t BufferUsed;
|
||||
Dispatcher *Dispatch;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *State;
|
||||
|
||||
uint32_t AllocateTmpSpace(size_t Size);
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, IR::OrderedNodeWrapper Op);
|
||||
|
||||
template<typename Res>
|
||||
Res GetSrc(void* SSAData, IR::OrderedNodeWrapper Src);
|
||||
|
||||
Dispatcher *Dispatcher{};
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
T AtomicCompareAndSwap(T expected, T desired, T *addr);
|
||||
|
||||
uint8_t AtomicFetchNeg(uint8_t *Addr);
|
||||
uint16_t AtomicFetchNeg(uint16_t *Addr);
|
||||
uint32_t AtomicFetchNeg(uint32_t *Addr);
|
||||
uint64_t AtomicFetchNeg(uint64_t *Addr);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
}
|
||||
@@ -1,109 +1,128 @@
|
||||
#include "Common/MathUtils.h"
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/DebugData.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <memory>
|
||||
#include <signal.h>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <atomic>
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
#include <vector>
|
||||
|
||||
#include "InterpreterOps.h"
|
||||
|
||||
#if defined(_M_X86_64)
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#elif defined(_M_ARM_64)
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
#else
|
||||
#error missing arch
|
||||
#endif
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
}
|
||||
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Dispatch(Dispatcher)
|
||||
{
|
||||
static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
|
||||
auto LocalEntry = Thread->LocalIRCache.find(Thread->CurrentFrame->State.rip);
|
||||
|
||||
Interpreter.FragmentExecuter = reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR);
|
||||
|
||||
ClearCache();
|
||||
InterpreterOps::InterpretIR(Thread, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
|
||||
}
|
||||
|
||||
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
bool InterpreterCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
|
||||
}, true);
|
||||
constexpr bool is_arm64 = true;
|
||||
#else
|
||||
constexpr bool is_arm64 = false;
|
||||
#endif
|
||||
}
|
||||
|
||||
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
const auto IRSize = AlignUp(IR->GetInlineSize(), 16);
|
||||
const auto MaxSize = IRSize + Dispatcher::MaxInterpreterTrampolineSize + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
|
||||
if ((BufferUsed + MaxSize) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState);
|
||||
if constexpr (is_arm64) {
|
||||
uint32_t *PC = reinterpret_cast<uint32_t*>(ArchHelpers::Context::GetPc(ucontext));
|
||||
uint32_t Instr = PC[0];
|
||||
if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS CASPAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS CASAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x: PC: %p Instruction: 0x%08x\n", Op, PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto BufferStart = CurrentCodeBuffer->Ptr + BufferUsed;
|
||||
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
: CTX {ctx}
|
||||
, State {Thread} {
|
||||
// Grab our space for temporary data
|
||||
|
||||
auto DestBuffer = BufferStart;
|
||||
if (!CompileThread &&
|
||||
CTX->Config.Core == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
CreateAsmDispatch(ctx, Thread);
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
});
|
||||
|
||||
if (GDBEnabled) {
|
||||
const auto GDBSize = Dispatch->GenerateGDBPauseCheck(DestBuffer, Entry);
|
||||
DestBuffer += GDBSize;
|
||||
BufferUsed += GDBSize;
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->HandleSIGBUS(Signal, info, ucontext);
|
||||
});
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
}
|
||||
|
||||
const auto TrampolineSize = Dispatch->GenerateInterpreterTrampoline(DestBuffer);
|
||||
DestBuffer += TrampolineSize;
|
||||
BufferUsed += TrampolineSize;
|
||||
|
||||
|
||||
IR->Serialize(DestBuffer);
|
||||
DestBuffer += IRSize;
|
||||
BufferUsed += IRSize;
|
||||
|
||||
return BufferStart;
|
||||
}
|
||||
|
||||
void InterpreterCore::ClearCache() {
|
||||
// Calling this one is needed to setup the initial CurrentCodeBuffer
|
||||
[[maybe_unused]] auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
BufferUsed = 0;
|
||||
|
||||
InterpreterCore::~InterpreterCore() {
|
||||
delete Dispatcher;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
|
||||
|
||||
void *InterpreterCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
return reinterpret_cast<void*>(InterpreterExecution);
|
||||
}
|
||||
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
InterpreterCore::InitializeSignalHandlers(CTX);
|
||||
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return new InterpreterCore(ctx, Thread, CompileThread);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,5 @@
|
||||
#pragma once
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
@@ -12,11 +10,7 @@ namespace FEXCore::Core {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
struct DispatcherConfig;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures();
|
||||
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
}
|
||||
@@ -1,184 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#define GD *GetDest<uint64_t*>(Data->SSAData, Node)
|
||||
#define GDP GetDest<void*>(Data->SSAData, Node)
|
||||
|
||||
#define DO_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(GDP); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
*Dst_d = func(*Src1_d, *Src2_d); \
|
||||
break; \
|
||||
}
|
||||
#define DO_SCALAR_COMPARE_OP(size, type, type2, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
Dst_d[0] = func(Src1_d[0], Src2_d[0]); \
|
||||
break; \
|
||||
}
|
||||
|
||||
#define DO_VECTOR_COMPARE_OP(size, type, type2, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_PAIR_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i*2], Src1_d[i*2 + 1]); \
|
||||
Dst_d[i+Elements] = func(Src2_d[i*2], Src2_d[i*2 + 1]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_SCALAR_OP(size, type, func)\
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i], *Src2_d); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_0SRC_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_1SRC_OP(size, type, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type*>(Src); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src_d[i]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_REDUCE_1SRC_OP(size, type, func, start_val) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type*>(Src); \
|
||||
type begin = start_val; \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
begin = func(begin, Src_d[i]); \
|
||||
} \
|
||||
Dst_d[0] = begin; \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_SAT_OP(size, type, func, min, max) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = func(Src1_d[i], Src2_d[i], min, max); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
|
||||
#define DO_VECTOR_1SRC_2TYPE_OP(size, type, type2, func, min, max) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type2*>(Src); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func(Src_d[i], min, max); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
|
||||
#define DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(type, type2, func, min, max) \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type2*>(Src); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func(Src_d[i], min, max); \
|
||||
}
|
||||
#define DO_VECTOR_1SRC_2TYPE_OP_TOP(size, type, type2, func, min, max) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type2*>(Src2); \
|
||||
memcpy(Dst_d, Src1, Elements * sizeof(type2));\
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i+Elements] = (type)func(Src_d[i], min, max); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
|
||||
#define DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(size, type, type2, func, min, max) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src_d = reinterpret_cast<type2*>(Src); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func(Src_d[i+Elements], min, max); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_2SRC_2TYPE_OP(size, type, type2, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func((type)Src1_d[i], (type)Src2_d[i]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
#define DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(size, type, type2, func) \
|
||||
case size: { \
|
||||
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
|
||||
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
|
||||
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
|
||||
for (uint8_t i = 0; i < Elements; ++i) { \
|
||||
Dst_d[i] = (type)func((type)Src1_d[i+Elements], (type)Src2_d[i+Elements]); \
|
||||
} \
|
||||
break; \
|
||||
}
|
||||
|
||||
struct InterpVector256 {
|
||||
__uint128_t Lower;
|
||||
__uint128_t Upper;
|
||||
};
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, FEXCore::IR::OrderedNodeWrapper Op) {
|
||||
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.ID().Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
|
||||
template<typename Res>
|
||||
Res GetDest(void* SSAData, FEXCore::IR::NodeID Op) {
|
||||
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Op.Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
|
||||
|
||||
template<typename Res>
|
||||
Res GetSrc(void* SSAData, FEXCore::IR::OrderedNodeWrapper Src) {
|
||||
auto DstPtr = &reinterpret_cast<InterpVector256*>(SSAData)[Src.ID().Value];
|
||||
return reinterpret_cast<Res>(DstPtr);
|
||||
}
|
||||
@@ -1,313 +0,0 @@
|
||||
#include "FEXCore/Core/CoreState.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/F80Ops.h"
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_UNKNOWN, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F32, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I16, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_VOID_U16, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I32, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F32_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I16_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I32_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I64_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I64_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F80LOADFCW] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW).fn);
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4).fn);
|
||||
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8).fn);
|
||||
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4).fn);
|
||||
Info[Core::OPINDEX_F80CVT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_TRUNC2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_TRUNC4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4).fn);
|
||||
Info[Core::OPINDEX_F80CVTINT_TRUNC8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8).fn);
|
||||
Info[Core::OPINDEX_F80CMP_0] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>, Core::OPINDEX_F80CMP_0).fn);
|
||||
Info[Core::OPINDEX_F80CMP_1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>, Core::OPINDEX_F80CMP_1).fn);
|
||||
Info[Core::OPINDEX_F80CMP_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>, Core::OPINDEX_F80CMP_2).fn);
|
||||
Info[Core::OPINDEX_F80CMP_3] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>, Core::OPINDEX_F80CMP_3).fn);
|
||||
Info[Core::OPINDEX_F80CMP_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>, Core::OPINDEX_F80CMP_4).fn);
|
||||
Info[Core::OPINDEX_F80CMP_5] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>, Core::OPINDEX_F80CMP_5).fn);
|
||||
Info[Core::OPINDEX_F80CMP_6] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>, Core::OPINDEX_F80CMP_6).fn);
|
||||
Info[Core::OPINDEX_F80CMP_7] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>, Core::OPINDEX_F80CMP_7).fn);
|
||||
Info[Core::OPINDEX_F80CVTTOINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2).fn);
|
||||
Info[Core::OPINDEX_F80CVTTOINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4).fn);
|
||||
|
||||
// Unary
|
||||
Info[Core::OPINDEX_F80ROUND] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ROUND>::handle, Core::OPINDEX_F80ROUND).fn);
|
||||
Info[Core::OPINDEX_F80F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80F2XM1>::handle, Core::OPINDEX_F80F2XM1).fn);
|
||||
Info[Core::OPINDEX_F80TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80TAN>::handle, Core::OPINDEX_F80TAN).fn);
|
||||
Info[Core::OPINDEX_F80SQRT] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SQRT>::handle, Core::OPINDEX_F80SQRT).fn);
|
||||
Info[Core::OPINDEX_F80SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SIN>::handle, Core::OPINDEX_F80SIN).fn);
|
||||
Info[Core::OPINDEX_F80COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80COS>::handle, Core::OPINDEX_F80COS).fn);
|
||||
Info[Core::OPINDEX_F80XTRACT_EXP] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle, Core::OPINDEX_F80XTRACT_EXP).fn);
|
||||
Info[Core::OPINDEX_F80XTRACT_SIG] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle, Core::OPINDEX_F80XTRACT_SIG).fn);
|
||||
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle, Core::OPINDEX_F80BCDSTORE).fn);
|
||||
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle, Core::OPINDEX_F80BCDLOAD).fn);
|
||||
|
||||
// Binary
|
||||
Info[Core::OPINDEX_F80ADD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ADD>::handle, Core::OPINDEX_F80ADD).fn);
|
||||
Info[Core::OPINDEX_F80SUB] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SUB>::handle, Core::OPINDEX_F80SUB).fn);
|
||||
Info[Core::OPINDEX_F80MUL] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80MUL>::handle, Core::OPINDEX_F80MUL).fn);
|
||||
Info[Core::OPINDEX_F80DIV] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80DIV>::handle, Core::OPINDEX_F80DIV).fn);
|
||||
Info[Core::OPINDEX_F80FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2X>::handle, Core::OPINDEX_F80FYL2X).fn);
|
||||
Info[Core::OPINDEX_F80ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ATAN>::handle, Core::OPINDEX_F80ATAN).fn);
|
||||
Info[Core::OPINDEX_F80FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM1>::handle, Core::OPINDEX_F80FPREM1).fn);
|
||||
Info[Core::OPINDEX_F80FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM>::handle, Core::OPINDEX_F80FPREM).fn);
|
||||
Info[Core::OPINDEX_F80SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle, Core::OPINDEX_F80SCALE).fn);
|
||||
|
||||
// Double Precision
|
||||
Info[Core::OPINDEX_F64SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle, Core::OPINDEX_F64SIN).fn);
|
||||
Info[Core::OPINDEX_F64COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle, Core::OPINDEX_F64COS).fn);
|
||||
Info[Core::OPINDEX_F64TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle, Core::OPINDEX_F64TAN).fn);
|
||||
Info[Core::OPINDEX_F64ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle, Core::OPINDEX_F64ATAN).fn);
|
||||
Info[Core::OPINDEX_F64F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle, Core::OPINDEX_F64F2XM1).fn);
|
||||
Info[Core::OPINDEX_F64FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle, Core::OPINDEX_F64FYL2X).fn);
|
||||
Info[Core::OPINDEX_F64FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle, Core::OPINDEX_F64FPREM).fn);
|
||||
Info[Core::OPINDEX_F64FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle, Core::OPINDEX_F64FPREM1).fn);
|
||||
Info[Core::OPINDEX_F64SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle, Core::OPINDEX_F64SCALE).fn);
|
||||
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVTINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
if (Op->Truncate) {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2);
|
||||
}
|
||||
else {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
if (Op->Truncate) {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4);
|
||||
}
|
||||
else {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
if (Op->Truncate) {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8);
|
||||
}
|
||||
else {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
static constexpr std::array handlers{
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = GetFallbackInfo(handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags));
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTOINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_X87_OP(ROUND)
|
||||
COMMON_X87_OP(F2XM1)
|
||||
COMMON_X87_OP(TAN)
|
||||
COMMON_X87_OP(SQRT)
|
||||
COMMON_X87_OP(SIN)
|
||||
COMMON_X87_OP(COS)
|
||||
COMMON_X87_OP(XTRACT_EXP)
|
||||
COMMON_X87_OP(XTRACT_SIG)
|
||||
COMMON_X87_OP(BCDSTORE)
|
||||
COMMON_X87_OP(BCDLOAD)
|
||||
|
||||
// Binary
|
||||
COMMON_X87_OP(ADD)
|
||||
COMMON_X87_OP(SUB)
|
||||
COMMON_X87_OP(MUL)
|
||||
COMMON_X87_OP(DIV)
|
||||
COMMON_X87_OP(FYL2X)
|
||||
COMMON_X87_OP(ATAN)
|
||||
COMMON_X87_OP(FPREM1)
|
||||
COMMON_X87_OP(FPREM)
|
||||
COMMON_X87_OP(SCALE)
|
||||
|
||||
// Double Precision Unary
|
||||
COMMON_F64_OP(F2XM1)
|
||||
COMMON_F64_OP(TAN)
|
||||
COMMON_F64_OP(SIN)
|
||||
COMMON_F64_OP(COS)
|
||||
|
||||
// Double Precision Binary
|
||||
COMMON_F64_OP(FYL2X)
|
||||
COMMON_F64_OP(ATAN)
|
||||
COMMON_F64_OP(FPREM1)
|
||||
COMMON_F64_OP(FPREM)
|
||||
COMMON_F64_OP(SCALE)
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
+5089
-333
File diff suppressed because it is too large.
Load diff
@@ -1,17 +1,9 @@
|
||||
#pragma once
|
||||
#include <stdint.h>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IRListView;
|
||||
struct IROp_Header;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core{
|
||||
@@ -28,8 +20,6 @@ namespace FEXCore::CPU {
|
||||
FABI_F80_I32,
|
||||
FABI_F32_F80,
|
||||
FABI_F64_F80,
|
||||
FABI_F64_F64,
|
||||
FABI_F64_F64_F64,
|
||||
FABI_I16_F80,
|
||||
FABI_I32_F80,
|
||||
FABI_I64_F80,
|
||||
@@ -41,376 +31,12 @@ namespace FEXCore::CPU {
|
||||
struct FallbackInfo {
|
||||
FallbackABI ABI;
|
||||
void *fn;
|
||||
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
|
||||
};
|
||||
|
||||
|
||||
class InterpreterOps {
|
||||
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR);
|
||||
static void FillFallbackIndexPointers(uint64_t *Info);
|
||||
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
|
||||
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
|
||||
|
||||
struct IROpData {
|
||||
FEXCore::Core::InternalThreadState *State{};
|
||||
uint64_t CurrentEntry{};
|
||||
FEXCore::IR::IRListView const *CurrentIR{};
|
||||
volatile void *StackEntry{};
|
||||
void *SSAData{};
|
||||
struct {
|
||||
bool Quit;
|
||||
bool Redo;
|
||||
} BlockResults{};
|
||||
|
||||
IR::NodeIterator BlockIterator{0, 0};
|
||||
};
|
||||
|
||||
#define DEF_OP(x) static void Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
|
||||
///< No-op Handler
|
||||
DEF_OP(NoOp);
|
||||
|
||||
///< ALU Ops
|
||||
DEF_OP(TruncElementPair);
|
||||
DEF_OP(Constant);
|
||||
DEF_OP(EntrypointOffset);
|
||||
DEF_OP(InlineConstant);
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
DEF_OP(UDiv);
|
||||
DEF_OP(Rem);
|
||||
DEF_OP(URem);
|
||||
DEF_OP(MulH);
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
DEF_OP(Lshl);
|
||||
DEF_OP(Lshr);
|
||||
DEF_OP(Ashr);
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
DEF_OP(LURem);
|
||||
DEF_OP(Zext);
|
||||
DEF_OP(Not);
|
||||
DEF_OP(Popcount);
|
||||
DEF_OP(FindLSB);
|
||||
DEF_OP(FindMSB);
|
||||
DEF_OP(FindTrailingZeros);
|
||||
DEF_OP(CountLeadingZeroes);
|
||||
DEF_OP(Rev);
|
||||
DEF_OP(Bfi);
|
||||
DEF_OP(Bfe);
|
||||
DEF_OP(Sbfe);
|
||||
DEF_OP(Select);
|
||||
DEF_OP(VExtractToGPR);
|
||||
DEF_OP(Float_ToGPR_ZU);
|
||||
DEF_OP(Float_ToGPR_ZS);
|
||||
DEF_OP(Float_ToGPR_S);
|
||||
DEF_OP(FCmp);
|
||||
|
||||
///< Atomic ops
|
||||
DEF_OP(CASPair);
|
||||
DEF_OP(CAS);
|
||||
DEF_OP(AtomicAdd);
|
||||
DEF_OP(AtomicSub);
|
||||
DEF_OP(AtomicAnd);
|
||||
DEF_OP(AtomicOr);
|
||||
DEF_OP(AtomicXor);
|
||||
DEF_OP(AtomicSwap);
|
||||
DEF_OP(AtomicFetchAdd);
|
||||
DEF_OP(AtomicFetchSub);
|
||||
DEF_OP(AtomicFetchAnd);
|
||||
DEF_OP(AtomicFetchOr);
|
||||
DEF_OP(AtomicFetchXor);
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
DEF_OP(Jump);
|
||||
DEF_OP(CondJump);
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
DEF_OP(Vector_FToZS);
|
||||
DEF_OP(Vector_FToS);
|
||||
DEF_OP(Vector_FToF);
|
||||
DEF_OP(Vector_FToI);
|
||||
|
||||
///< Flag ops
|
||||
DEF_OP(GetHostFlag);
|
||||
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
DEF_OP(FillRegister);
|
||||
DEF_OP(LoadFlag);
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Phi);
|
||||
DEF_OP(PhiValue);
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
DEF_OP(Mov);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
DEF_OP(VOr);
|
||||
DEF_OP(VXor);
|
||||
DEF_OP(VAdd);
|
||||
DEF_OP(VSub);
|
||||
DEF_OP(VUQAdd);
|
||||
DEF_OP(VUQSub);
|
||||
DEF_OP(VSQAdd);
|
||||
DEF_OP(VSQSub);
|
||||
DEF_OP(VAddP);
|
||||
DEF_OP(VAddV);
|
||||
DEF_OP(VUMinV);
|
||||
DEF_OP(VURAvg);
|
||||
DEF_OP(VAbs);
|
||||
DEF_OP(VPopcount);
|
||||
DEF_OP(VFAdd);
|
||||
DEF_OP(VFAddP);
|
||||
DEF_OP(VFSub);
|
||||
DEF_OP(VFMul);
|
||||
DEF_OP(VFDiv);
|
||||
DEF_OP(VFMin);
|
||||
DEF_OP(VFMax);
|
||||
DEF_OP(VFRecp);
|
||||
DEF_OP(VFSqrt);
|
||||
DEF_OP(VFRSqrt);
|
||||
DEF_OP(VNeg);
|
||||
DEF_OP(VFNeg);
|
||||
DEF_OP(VNot);
|
||||
DEF_OP(VUMin);
|
||||
DEF_OP(VSMin);
|
||||
DEF_OP(VUMax);
|
||||
DEF_OP(VSMax);
|
||||
DEF_OP(VZip);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
DEF_OP(VCMPGT);
|
||||
DEF_OP(VCMPGTZ);
|
||||
DEF_OP(VCMPLTZ);
|
||||
DEF_OP(VFCMPEQ);
|
||||
DEF_OP(VFCMPNEQ);
|
||||
DEF_OP(VFCMPLT);
|
||||
DEF_OP(VFCMPGT);
|
||||
DEF_OP(VFCMPLE);
|
||||
DEF_OP(VFCMPORD);
|
||||
DEF_OP(VFCMPUNO);
|
||||
DEF_OP(VUShl);
|
||||
DEF_OP(VUShr);
|
||||
DEF_OP(VSShr);
|
||||
DEF_OP(VUShlS);
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
DEF_OP(VUXTL2);
|
||||
DEF_OP(VSQXTN);
|
||||
DEF_OP(VSQXTN2);
|
||||
DEF_OP(VSQXTUN);
|
||||
DEF_OP(VSQXTUN2);
|
||||
DEF_OP(VUMul);
|
||||
DEF_OP(VUMull);
|
||||
DEF_OP(VSMul);
|
||||
DEF_OP(VSMull);
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
DEF_OP(AESEnc);
|
||||
DEF_OP(AESEncLast);
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
|
||||
///< F80 ops
|
||||
DEF_OP(F80LOADFCW);
|
||||
DEF_OP(F80ADD);
|
||||
DEF_OP(F80SUB);
|
||||
DEF_OP(F80MUL);
|
||||
DEF_OP(F80DIV);
|
||||
DEF_OP(F80FYL2X);
|
||||
DEF_OP(F80ATAN);
|
||||
DEF_OP(F80FPREM1);
|
||||
DEF_OP(F80FPREM);
|
||||
DEF_OP(F80SCALE);
|
||||
DEF_OP(F80CVT);
|
||||
DEF_OP(F80CVTINT);
|
||||
DEF_OP(F80CVTTO);
|
||||
DEF_OP(F80CVTTOINT);
|
||||
DEF_OP(F80ROUND);
|
||||
DEF_OP(F80F2XM1);
|
||||
DEF_OP(F80TAN);
|
||||
DEF_OP(F80SQRT);
|
||||
DEF_OP(F80SIN);
|
||||
DEF_OP(F80COS);
|
||||
DEF_OP(F80XTRACT_EXP);
|
||||
DEF_OP(F80XTRACT_SIG);
|
||||
DEF_OP(F80CMP);
|
||||
DEF_OP(F80BCDLOAD);
|
||||
DEF_OP(F80BCDSTORE);
|
||||
|
||||
//< F64 ops
|
||||
DEF_OP(F64SIN);
|
||||
DEF_OP(F64COS);
|
||||
DEF_OP(F64TAN);
|
||||
DEF_OP(F64F2XM1);
|
||||
DEF_OP(F64ATAN);
|
||||
DEF_OP(F64FPREM);
|
||||
DEF_OP(F64FPREM1);
|
||||
DEF_OP(F64FYL2X);
|
||||
DEF_OP(F64SCALE);
|
||||
#undef DEF_OP
|
||||
template<typename unsigned_type, typename signed_type, typename float_type>
|
||||
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
|
||||
bool CompResult = false;
|
||||
switch (Cond) {
|
||||
case FEXCore::IR::COND_EQ:
|
||||
CompResult = static_cast<unsigned_type>(Src1) == static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_NEQ:
|
||||
CompResult = static_cast<unsigned_type>(Src1) != static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_SGE:
|
||||
CompResult = static_cast<signed_type>(Src1) >= static_cast<signed_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_SLT:
|
||||
CompResult = static_cast<signed_type>(Src1) < static_cast<signed_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_SGT:
|
||||
CompResult = static_cast<signed_type>(Src1) > static_cast<signed_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_SLE:
|
||||
CompResult = static_cast<signed_type>(Src1) <= static_cast<signed_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_UGE:
|
||||
CompResult = static_cast<unsigned_type>(Src1) >= static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_ULT:
|
||||
CompResult = static_cast<unsigned_type>(Src1) < static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_UGT:
|
||||
CompResult = static_cast<unsigned_type>(Src1) > static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
case FEXCore::IR::COND_ULE:
|
||||
CompResult = static_cast<unsigned_type>(Src1) <= static_cast<unsigned_type>(Src2);
|
||||
break;
|
||||
|
||||
case FEXCore::IR::COND_FLU:
|
||||
CompResult = reinterpret_cast<float_type&>(Src1) < reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FGE:
|
||||
CompResult = reinterpret_cast<float_type&>(Src1) >= reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FLEU:
|
||||
CompResult = reinterpret_cast<float_type&>(Src1) <= reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FGT:
|
||||
CompResult = reinterpret_cast<float_type&>(Src1) > reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FU:
|
||||
CompResult = (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_FNU:
|
||||
CompResult = !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
|
||||
break;
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
break;
|
||||
}
|
||||
|
||||
return CompResult;
|
||||
}
|
||||
|
||||
static uint8_t GetOpSize(FEXCore::IR::IRListView const *CurrentIR, IR::OrderedNodeWrapper Node) {
|
||||
auto IROp = CurrentIR->GetOp<FEXCore::IR::IROp_Header>(Node);
|
||||
return IROp->Size;
|
||||
}
|
||||
|
||||
};
|
||||
} // namespace FEXCore::CPU
|
||||
};
|
||||
@@ -1,286 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static inline void CacheLineFlush(char *Addr) {
|
||||
#ifdef _M_X86_64
|
||||
__asm volatile (
|
||||
"clflush (%[Addr]);"
|
||||
:: [Addr] "r" (Addr)
|
||||
: "memory");
|
||||
#else
|
||||
__builtin___clear_cache(Addr, Addr+64);
|
||||
#endif
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
switch (IROp->Size) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, IROp->Size);
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(FillRegister) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
|
||||
|
||||
switch(Op->OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
|
||||
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
memset(GDP, 0, 16);
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint16_t>*>(MemData);
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint32_t>*>(MemData);
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint64_t>*>(MemData);
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
|
||||
|
||||
switch(Op->OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
|
||||
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
reinterpret_cast<std::atomic<uint8_t>*>(MemData)->store(*GetSrc<uint8_t*>(Data->SSAData, Op->Value));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
reinterpret_cast<std::atomic<uint16_t>*>(MemData)->store(*GetSrc<uint16_t*>(Data->SSAData, Op->Value));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
reinterpret_cast<std::atomic<uint32_t>*>(MemData)->store(*GetSrc<uint32_t*>(Data->SSAData, Op->Value));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
reinterpret_cast<std::atomic<uint64_t>*>(MemData)->store(*GetSrc<uint64_t*>(Data->SSAData, Op->Value));
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), IROp->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Value);
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Addr), 16);
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
|
||||
MemData, Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(VStoreMemElement) {
|
||||
#define STORE_DATA(x, y) \
|
||||
case x: { \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Value); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Addr)[Op->Index], sizeof(y)); \
|
||||
break; \
|
||||
}
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VStoreMemElement>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
STORE_DATA(1, uint8_t)
|
||||
STORE_DATA(2, uint16_t)
|
||||
STORE_DATA(4, uint32_t)
|
||||
STORE_DATA(8, uint64_t)
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size"); break;
|
||||
}
|
||||
#undef STORE_DATA
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
char *MemData = *GetSrc<char **>(Data->SSAData, Op->Addr);
|
||||
|
||||
// 64-byte cache line clear
|
||||
CacheLineFlush(MemData);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
uintptr_t MemData = *GetSrc<uintptr_t*>(Data->SSAData, Op->Addr);
|
||||
|
||||
// Force cacheline alignment
|
||||
MemData = MemData & ~(CPUIDEmu::CACHELINE_SIZE - 1);
|
||||
|
||||
using DataType = uint64_t;
|
||||
DataType *MemData64 = reinterpret_cast<DataType*>(MemData);
|
||||
|
||||
// 64-byte cache line zero
|
||||
for (size_t i = 0; i < (CPUIDEmu::CACHELINE_SIZE / sizeof(DataType)); ++i) {
|
||||
MemData64[i] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,181 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cstdint>
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#endif
|
||||
#include <sys/random.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
static void StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->StopThread(Thread);
|
||||
|
||||
LOGMAN_MSG_A_FMT("unreachable");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
case IR::Fence_Load.Val:
|
||||
std::atomic_thread_fence(std::memory_order_acquire);
|
||||
break;
|
||||
case IR::Fence_LoadStore.Val:
|
||||
std::atomic_thread_fence(std::memory_order_seq_cst);
|
||||
break;
|
||||
case IR::Fence_Store.Val:
|
||||
std::atomic_thread_fence(std::memory_order_release);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
|
||||
Data->State->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = 1;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.Signal = Op->Reason.Signal;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.TrapNo = Op->Reason.TrapNumber;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.err_code = Op->Reason.ErrorRegister;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.si_code = Op->Reason.si_code;
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGSEGV);
|
||||
break;
|
||||
default:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(GetRoundingMode) {
|
||||
uint32_t GuestRounding{};
|
||||
#ifdef _M_ARM_64
|
||||
uint64_t Tmp{};
|
||||
__asm(R"(
|
||||
mrs %[Tmp], FPCR;
|
||||
)"
|
||||
: [Tmp] "=r" (Tmp));
|
||||
// Extract the rounding
|
||||
// On ARM the ordering is different than on x86
|
||||
GuestRounding |= ((Tmp >> 24) & 1) ? IR::ROUND_MODE_FLUSH_TO_ZERO : 0;
|
||||
uint8_t RoundingMode = (Tmp >> 22) & 0b11;
|
||||
if (RoundingMode == 0)
|
||||
GuestRounding |= IR::ROUND_MODE_NEAREST;
|
||||
else if (RoundingMode == 1)
|
||||
GuestRounding |= IR::ROUND_MODE_POSITIVE_INFINITY;
|
||||
else if (RoundingMode == 2)
|
||||
GuestRounding |= IR::ROUND_MODE_NEGATIVE_INFINITY;
|
||||
else if (RoundingMode == 3)
|
||||
GuestRounding |= IR::ROUND_MODE_TOWARDS_ZERO;
|
||||
#else
|
||||
GuestRounding = _mm_getcsr();
|
||||
|
||||
// Extract the rounding
|
||||
GuestRounding = (GuestRounding >> 13) & 0b111;
|
||||
#endif
|
||||
memcpy(GDP, &GuestRounding, sizeof(GuestRounding));
|
||||
}
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
const auto GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->RoundMode);
|
||||
#ifdef _M_ARM_64
|
||||
uint64_t HostRounding{};
|
||||
__asm volatile(R"(
|
||||
mrs %[Tmp], FPCR;
|
||||
)"
|
||||
: [Tmp] "=r" (HostRounding));
|
||||
// Mask out the rounding
|
||||
HostRounding &= ~(0b111 << 22);
|
||||
|
||||
HostRounding |= (GuestRounding & IR::ROUND_MODE_FLUSH_TO_ZERO) ? (1U << 24) : 0;
|
||||
|
||||
uint8_t RoundingMode = GuestRounding & 0b11;
|
||||
if (RoundingMode == IR::ROUND_MODE_NEAREST)
|
||||
HostRounding |= (0b00U << 22);
|
||||
else if (RoundingMode == IR::ROUND_MODE_POSITIVE_INFINITY)
|
||||
HostRounding |= (0b01U << 22);
|
||||
else if (RoundingMode == IR::ROUND_MODE_NEGATIVE_INFINITY)
|
||||
HostRounding |= (0b10U << 22);
|
||||
else if (RoundingMode == IR::ROUND_MODE_TOWARDS_ZERO)
|
||||
HostRounding |= (0b11U << 22);
|
||||
|
||||
__asm volatile(R"(
|
||||
msr FPCR, %[Tmp];
|
||||
)"
|
||||
:: [Tmp] "r" (HostRounding));
|
||||
#else
|
||||
uint32_t HostRounding = _mm_getcsr();
|
||||
|
||||
// Cut out the host rounding mode
|
||||
HostRounding &= ~(0b111 << 13);
|
||||
|
||||
// Insert our new rounding mode
|
||||
HostRounding |= GuestRounding << 13;
|
||||
_mm_setcsr(HostRounding);
|
||||
#endif
|
||||
}
|
||||
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
if (OpSize <= 8) {
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
LogMan::Msg::IFmt(">>>> Value in Arg: 0x{:x}, {}", Src, Src);
|
||||
}
|
||||
else if (OpSize == 16) {
|
||||
const auto Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Value);
|
||||
const uint64_t Src0 = Src;
|
||||
const uint64_t Src1 = Src >> 64;
|
||||
LogMan::Msg::IFmt(">>>> Value[0] in Arg: 0x{:x}, {}", Src0, Src0);
|
||||
LogMan::Msg::IFmt(" Value[1] in Arg: 0x{:x}, {}", Src1, Src1);
|
||||
}
|
||||
else
|
||||
LOGMAN_MSG_A_FMT("Unknown value size: {}", OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
uint32_t CPU, CPUNode;
|
||||
FHU::Syscalls::getcpu(&CPU, &CPUNode);
|
||||
GD = (CPUNode << 12) | CPU;
|
||||
}
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
// We are ignoring Op->GetReseeded in the interpreter
|
||||
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
ssize_t Result = ::getrandom(&DstPtr[0], 8, 0);
|
||||
|
||||
// Second result is if we managed to read a valid random number or not
|
||||
DstPtr[1] = Result == 8 ? 1 : 0;
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
// Nop implementation
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,42 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
const auto Src = GetSrc<uintptr_t>(Data->SSAData, Op->Pair);
|
||||
memcpy(GDP,
|
||||
reinterpret_cast<void*>(Src + Op->Header.Size * Op->Element), Op->Header.Size);
|
||||
}
|
||||
|
||||
DEF_OP(CreateElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_CreateElementPair>();
|
||||
const void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Lower);
|
||||
const void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Upper);
|
||||
|
||||
uint8_t *Dst = GetDest<uint8_t*>(Data->SSAData, Node);
|
||||
|
||||
memcpy(Dst, Src_Lower, IROp->ElementSize);
|
||||
memcpy(Dst + IROp->ElementSize, Src_Upper, IROp->ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
File diff suppressed because it is too large.
Load diff
+306
-542
File diff suppressed because it is too large.
Load diff
@@ -1,131 +0,0 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|arm64
|
||||
desc: relocation logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
break;
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Lit {
|
||||
.Lit = Literal(Pointer),
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
|
||||
|
||||
place(&Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
Literal<uint64_t> Lit(Pointer);
|
||||
place(&Lit);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
+165
-214
@@ -4,26 +4,26 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
// Size is the size of each pair element
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
auto Expected = GetSrcPair<RA_64>(Op->Expected.ID());
|
||||
auto Desired = GetSrcPair<RA_64>(Op->Desired.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto Expected = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Desired = GetSrcPair<RA_64>(Op->Header.Args[1].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
if (SupportsAtomics) {
|
||||
mov(TMP3, Expected.first);
|
||||
mov(TMP4, Expected.second);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
caspal(TMP3.W(), TMP4.W(), Desired.first.W(), Desired.second.W(), MemOperand(MemSrc));
|
||||
mov(Dst.first.W(), TMP3.W());
|
||||
@@ -34,17 +34,16 @@ DEF_OP(CASPair) {
|
||||
mov(Dst.first, TMP3);
|
||||
mov(Dst.second, TMP4);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
|
||||
default: LogMan::Msg::A("Unsupported: %d", OpSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (IROp->ElementSize) {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
aarch64::Label LoopTop;
|
||||
aarch64::Label LoopNotExpected;
|
||||
aarch64::Label LoopExpected;
|
||||
bind(&LoopTop);
|
||||
|
||||
ldaxp(TMP2.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cmp(TMP2.W(), Expected.first.W());
|
||||
ccmp(TMP3.W(), Expected.second.W(), NoFlag, Condition::eq);
|
||||
@@ -70,7 +69,6 @@ DEF_OP(CASPair) {
|
||||
aarch64::Label LoopNotExpected;
|
||||
aarch64::Label LoopExpected;
|
||||
bind(&LoopTop);
|
||||
|
||||
ldaxp(TMP2.X(), TMP3.X(), MemOperand(MemSrc));
|
||||
cmp(TMP2.X(), Expected.first.X());
|
||||
ccmp(TMP3.X(), Expected.second.X(), NoFlag, Condition::eq);
|
||||
@@ -91,7 +89,7 @@ DEF_OP(CASPair) {
|
||||
bind(&LoopExpected);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
|
||||
default: LogMan::Msg::A("Unsupported: %d", OpSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -99,22 +97,25 @@ DEF_OP(CASPair) {
|
||||
DEF_OP(CAS) {
|
||||
auto Op = IROp->C<IR::IROp_CAS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
// Args[0]: Expected
|
||||
// Args[1]: Desired
|
||||
// Args[2]: Pointer
|
||||
// DataSrc = *Src1
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
|
||||
auto Expected = GetReg<RA_64>(Op->Expected.ID());
|
||||
auto Desired = GetReg<RA_64>(Op->Desired.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto Expected = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Desired = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
if (SupportsAtomics) {
|
||||
mov(TMP2, Expected);
|
||||
switch (OpSize) {
|
||||
case 1: casalb(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
|
||||
case 2: casalh(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
|
||||
case 4: casal(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
|
||||
case 8: casal(TMP2.X(), Desired.X(), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unsupported: %d", OpSize);
|
||||
}
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
}
|
||||
@@ -205,7 +206,7 @@ DEF_OP(CAS) {
|
||||
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", OpSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -213,25 +214,25 @@ DEF_OP(CAS) {
|
||||
DEF_OP(AtomicAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAdd>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: staddlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 2: staddlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 4: staddl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 8: staddl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
case 1: staddlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: staddlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 4: staddl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 8: staddl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -240,7 +241,7 @@ DEF_OP(AtomicAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -249,7 +250,7 @@ DEF_OP(AtomicAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -258,12 +259,12 @@ DEF_OP(AtomicAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
add(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
add(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -271,26 +272,26 @@ DEF_OP(AtomicAdd) {
|
||||
DEF_OP(AtomicSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSub>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
if (SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
case 1: staddlb(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 2: staddlh(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 4: staddl(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 8: staddl(TMP2.X(), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -299,7 +300,7 @@ DEF_OP(AtomicSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -308,7 +309,7 @@ DEF_OP(AtomicSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -317,12 +318,12 @@ DEF_OP(AtomicSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
sub(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
sub(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -330,26 +331,26 @@ DEF_OP(AtomicSub) {
|
||||
DEF_OP(AtomicAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAnd>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
if (SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
case 1: stclrlb(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 2: stclrlh(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 4: stclrl(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
case 8: stclrl(TMP2.X(), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -358,7 +359,7 @@ DEF_OP(AtomicAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -367,7 +368,7 @@ DEF_OP(AtomicAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -376,12 +377,12 @@ DEF_OP(AtomicAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
and_(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
and_(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -389,25 +390,25 @@ DEF_OP(AtomicAnd) {
|
||||
DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: stsetlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 2: stsetlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 4: stsetl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 8: stsetl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
case 1: stsetlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: stsetlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 4: stsetl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 8: stsetl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -416,7 +417,7 @@ DEF_OP(AtomicOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -425,7 +426,7 @@ DEF_OP(AtomicOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -434,12 +435,12 @@ DEF_OP(AtomicOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
orr(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
orr(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -447,25 +448,25 @@ DEF_OP(AtomicOr) {
|
||||
DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: steorlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 2: steorlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 4: steorl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
case 8: steorl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
case 1: steorlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: steorlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 4: steorl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 8: steorl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -474,7 +475,7 @@ DEF_OP(AtomicXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -483,7 +484,7 @@ DEF_OP(AtomicXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP2.W(), &LoopTop);
|
||||
break;
|
||||
@@ -492,12 +493,12 @@ DEF_OP(AtomicXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
eor(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
eor(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP2, TMP2, MemOperand(MemSrc));
|
||||
cbnz(TMP2, &LoopTop);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -505,44 +506,45 @@ DEF_OP(AtomicXor) {
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: swpalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: swpalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: swpal(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: swpal(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
if (SupportsAtomics) {
|
||||
mov(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
case 1: swplb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: swplh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: swpl(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: swpl(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
mov(TMP3, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
stlxrb(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
uxtb(GetReg<RA_32>(Node), TMP2.W());
|
||||
uxtb(GetReg<RA_64>(Node), TMP2.W());
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
stlxrh(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
uxtw(GetReg<RA_32>(Node), TMP2.W());
|
||||
uxtw(GetReg<RA_64>(Node), TMP2.W());
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
stlxr(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
break;
|
||||
@@ -551,37 +553,37 @@ DEF_OP(AtomicSwap) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
stlxr(TMP4, GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc));
|
||||
stlxr(TMP4, TMP3.X(), MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2.X());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldaddalb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldaddalh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldaddal(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldaddal(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
case 1: ldaddalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldaddalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldaddal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldaddal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -591,7 +593,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -601,7 +603,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -611,39 +613,39 @@ DEF_OP(AtomicFetchAdd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
add(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
add(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
if (SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
case 1: ldaddalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldaddalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldaddal(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldaddal(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -653,7 +655,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -663,7 +665,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -673,39 +675,39 @@ DEF_OP(AtomicFetchSub) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
sub(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
sub(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
switch (IROp->Size) {
|
||||
if (SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (Op->Size) {
|
||||
case 1: ldclralb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldclralh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldclral(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldclral(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -715,7 +717,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -725,7 +727,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -735,38 +737,38 @@ DEF_OP(AtomicFetchAnd) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
and_(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
and_(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldsetalb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldsetalh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldsetal(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldsetal(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
case 1: ldsetalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldsetalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldsetal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldsetal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -776,7 +778,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -786,7 +788,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -796,38 +798,38 @@ DEF_OP(AtomicFetchOr) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
orr(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
orr(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldeoralb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldeoralh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldeoral(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldeoral(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
if (SupportsAtomics) {
|
||||
switch (Op->Size) {
|
||||
case 1: ldeoralb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldeoralh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 4: ldeoral(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 8: ldeoral(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -837,7 +839,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -847,7 +849,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
|
||||
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
@@ -857,67 +859,17 @@ DEF_OP(AtomicFetchXor) {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
eor(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
|
||||
eor(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicFetchNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
// TMP2-TMP3
|
||||
switch (IROp->Size) {
|
||||
case 1: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrb(TMP2.W(), MemOperand(MemSrc));
|
||||
neg(TMP3.W(), TMP2.W());
|
||||
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxrh(TMP2.W(), MemOperand(MemSrc));
|
||||
neg(TMP3.W(), TMP2.W());
|
||||
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2.W(), MemOperand(MemSrc));
|
||||
neg(TMP3.W(), TMP2.W());
|
||||
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
|
||||
cbnz(TMP4.W(), &LoopTop);
|
||||
mov(GetReg<RA_32>(Node), TMP2.W());
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
aarch64::Label LoopTop;
|
||||
bind(&LoopTop);
|
||||
ldaxr(TMP2, MemOperand(MemSrc));
|
||||
neg(TMP3, TMP2);
|
||||
stlxr(TMP4, TMP3, MemOperand(MemSrc));
|
||||
cbnz(TMP4, &LoopTop);
|
||||
mov(GetReg<RA_64>(Node), TMP2);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterAtomicHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -934,7 +886,6 @@ void Arm64JITCore::RegisterAtomicHandlers() {
|
||||
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
|
||||
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
|
||||
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+96
-218
@@ -4,22 +4,28 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::D("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::D("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestReturn) {
|
||||
LogMan::Msg::D("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// First we must reset the stack
|
||||
@@ -27,7 +33,7 @@ DEF_OP(SignalReturn) {
|
||||
|
||||
// Now branch to our signal return helper
|
||||
// This can't be a direct branch since the code needs to live at a constant location
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)));
|
||||
LoadConstant(x0, ThreadSharedData.SignalReturnInstruction);
|
||||
br(x0);
|
||||
}
|
||||
|
||||
@@ -40,10 +46,10 @@ DEF_OP(CallbackReturn) {
|
||||
ResetStack();
|
||||
|
||||
// We can now lower the ref counter again
|
||||
|
||||
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(ThreadSharedData.SignalHandlerRefCounterPtr));
|
||||
ldr(w2, MemOperand(x0));
|
||||
sub(w2, w2, 1);
|
||||
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
str(w2, MemOperand(x0));
|
||||
|
||||
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
@@ -67,7 +73,7 @@ DEF_OP(ExitFunction) {
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
Literal l_BranchHost{ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker};
|
||||
Literal l_BranchHost{Dispatcher->ExitFunctionLinkerAddress};
|
||||
Literal l_BranchGuest{NewRIP};
|
||||
|
||||
ldr(x0, &l_BranchHost);
|
||||
@@ -76,10 +82,10 @@ DEF_OP(ExitFunction) {
|
||||
place(&l_BranchHost);
|
||||
place(&l_BranchGuest);
|
||||
} else {
|
||||
RipReg = GetReg<RA_64>(Op->NewRIP.ID());
|
||||
RipReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
// L1 Cache
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)));
|
||||
LoadConstant(x0, ThreadState->LookupCache->GetL1Pointer());
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
@@ -90,23 +96,30 @@ DEF_OP(ExitFunction) {
|
||||
br(x1);
|
||||
|
||||
bind(&FullLookup);
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop)));
|
||||
LoadConstant(TMP1, Dispatcher->AbsoluteLoopTopAddress);
|
||||
str(RipReg, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
br(TMP1);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto Target = Op->TargetBlock.ID();
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
Label *TargetLabel;
|
||||
auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID());
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second;
|
||||
}
|
||||
else {
|
||||
TargetLabel = &IsTarget->second;
|
||||
}
|
||||
PendingTargetLabel = TargetLabel;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
#define GRFCMP(Node) (Op->CompareSize == 4 ? GetDst(Node).S() : GetDst(Node).D())
|
||||
|
||||
static Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return Condition::eq;
|
||||
case FEXCore::IR::COND_NEQ: return Condition::ne;
|
||||
@@ -121,7 +134,7 @@ static Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
case FEXCore::IR::COND_FLU: return Condition::lt;
|
||||
case FEXCore::IR::COND_FGE: return Condition::ge;
|
||||
case FEXCore::IR::COND_FLEU:return Condition::le;
|
||||
case FEXCore::IR::COND_FGT: return Condition::gt;
|
||||
case FEXCore::IR::COND_FGT: return Condition::hi;
|
||||
case FEXCore::IR::COND_FU: return Condition::vs;
|
||||
case FEXCore::IR::COND_FNU: return Condition::vc;
|
||||
case FEXCore::IR::COND_VS:
|
||||
@@ -129,7 +142,7 @@ static Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
LogMan::Msg::A("Unsupported compare type");
|
||||
return Condition::nv;
|
||||
}
|
||||
}
|
||||
@@ -138,34 +151,51 @@ static Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
Label *TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
Label *TrueTargetLabel;
|
||||
Label *FalseTargetLabel;
|
||||
|
||||
auto TrueIter = JumpTargets.find(Op->TrueBlock.ID());
|
||||
auto FalseIter = JumpTargets.find(Op->FalseBlock.ID());
|
||||
|
||||
if (TrueIter == JumpTargets.end()) {
|
||||
TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
}
|
||||
else {
|
||||
TrueTargetLabel = &TrueIter->second;
|
||||
}
|
||||
|
||||
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LogMan::Throw::A(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbz(GRCMP(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LogMan::Throw::A(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbnz(GRCMP(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else {
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
if (isConst) {
|
||||
if (isConst)
|
||||
cmp(GRCMP(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
else
|
||||
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
fcmp(GRFCMP(Op->Cmp1.ID()), GRFCMP(Op->Cmp2.ID()));
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("CondJump: Expected GPR or FPR");
|
||||
LogMan::Msg::A("CondJump: Expected GPR or FPR");
|
||||
}
|
||||
|
||||
b(TrueTargetLabel, MapBranchCC(Op->Cond));
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
if (FalseIter == JumpTargets.end()) {
|
||||
FalseTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
}
|
||||
else {
|
||||
FalseTargetLabel = &FalseIter->second;
|
||||
}
|
||||
PendingTargetLabel = FalseTargetLabel;
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
@@ -175,16 +205,8 @@ DEF_OP(Syscall) {
|
||||
// X1: ThreadState
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
|
||||
SpillStaticRegs();
|
||||
}
|
||||
else {
|
||||
// Need to spill all caller saved registers still
|
||||
SpillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
|
||||
}
|
||||
SpillStaticRegs();
|
||||
|
||||
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
|
||||
sub(sp, sp, SPOffset);
|
||||
@@ -193,178 +215,22 @@ DEF_OP(Syscall) {
|
||||
str(GetReg<RA_64>(Op->Header.Args[i].ID()), MemOperand(sp, i * 8));
|
||||
}
|
||||
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)));
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(CTX->SyscallHandler));
|
||||
mov(x1, STATE);
|
||||
mov(x2, sp);
|
||||
|
||||
LoadConstant(x3, reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall));
|
||||
blr(x3);
|
||||
|
||||
add(sp, sp, SPOffset);
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY &&
|
||||
(Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
FillStaticRegs();
|
||||
}
|
||||
else {
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
|
||||
}
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Move result to its destination register
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(InlineSyscall) {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
// Arguments are passed as follows:
|
||||
// X8: SyscallNumber - RA INTERSECT
|
||||
// X0: Arg0 & Return
|
||||
// X1: Arg1
|
||||
// X2: Arg2
|
||||
// X3: Arg3
|
||||
// X4: Arg4 - RA INTERSECT
|
||||
// X5: Arg5 - RA INTERSECT
|
||||
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
|
||||
|
||||
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
|
||||
const static std::array<vixl::aarch64::Register, FEXCore::HLE::SyscallArguments::MAX_ARGS-1> RegArgs = {{
|
||||
x0, x1, x2, x3, x4, x5
|
||||
}};
|
||||
|
||||
bool Intersects{};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
std::vector<vixl::aarch64::Register> IntersectRegs(FEXCore::HLE::SyscallArguments::MAX_ARGS);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
|
||||
if (Reg.GetCode() == x8.GetCode() ||
|
||||
Reg.GetCode() == x4.GetCode() ||
|
||||
Reg.GetCode() == x5.GetCode()) {
|
||||
|
||||
SpillMask |= (1U << Reg.GetCode());
|
||||
Intersects = true;
|
||||
}
|
||||
}
|
||||
// XXX: For some reason spilling only the x4, x5, and x8 registers was causing issues
|
||||
// Come back to this once investigation reveals why it fails the gvisor ioctl test
|
||||
// For now override to all GPRs
|
||||
SpillMask = ~0U;
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(x0, SpillMask & 0xFFFF);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
// Now that we have claimed to be a syscall we can set up the arguments
|
||||
if (Intersects) {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg.GetCode() == x8.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
|
||||
}
|
||||
else if (Reg.GetCode() == x4.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
|
||||
}
|
||||
else if (Reg.GetCode() == x5.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg.GetCode() == x8.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
|
||||
}
|
||||
else if (Reg.GetCode() == x4.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
|
||||
}
|
||||
else if (Reg.GetCode() == x5.GetCode()) {
|
||||
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
else {
|
||||
mov(RegArgs[i], Reg);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Reg = GetReg<RA_32>(Op->Header.Args[i].ID());
|
||||
if (Reg.GetCode() == w8.GetCode()) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
|
||||
}
|
||||
else if (Reg.GetCode() == w4.GetCode()) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
|
||||
}
|
||||
else if (Reg.GetCode() == w5.GetCode()) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
else {
|
||||
uxtw(RegArgs[i].W(), Reg);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
mov(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
|
||||
}
|
||||
else {
|
||||
uxtw(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LoadConstant(x8, Op->HostSyscallNumber);
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
// Result is now in x0
|
||||
// Move result to its destination register
|
||||
if (CTX->Config.Is64BitMode()) {
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
else {
|
||||
uxtw(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
}
|
||||
// Move result to its destination register
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
@@ -377,7 +243,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(x0, GetReg<RA_64>(Op->ArgPtr.ID()));
|
||||
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(x2, (uintptr_t)thunkFn);
|
||||
@@ -388,20 +254,21 @@ DEF_OP(Thunk) {
|
||||
FillStaticRegs(); // load from ctx after ra64 refill
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
const auto *OldCode = (const uint8_t *)&Op->CodeOriginalLow;
|
||||
uint8_t *OldCode = (uint8_t *)&Op->CodeOriginalLow;
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
LoadConstant(GetReg<RA_64>(Node), 0);
|
||||
LoadConstant(x0, Entry + Op->Offset);
|
||||
LoadConstant(x0, IR->GetHeader()->Entry + Op->Offset);
|
||||
LoadConstant(x1, 1);
|
||||
|
||||
while (len >= 8)
|
||||
{
|
||||
ldr(x2, MemOperand(x0, idx));
|
||||
LoadConstant(x3, *(const uint32_t *)(OldCode + idx));
|
||||
LoadConstant(x3, *(uint32_t *)(OldCode + idx));
|
||||
cmp(x2, x3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
len -= 8;
|
||||
@@ -410,7 +277,7 @@ DEF_OP(ValidateCode) {
|
||||
while (len >= 4)
|
||||
{
|
||||
ldr(w2, MemOperand(x0, idx));
|
||||
LoadConstant(w3, *(const uint32_t *)(OldCode + idx));
|
||||
LoadConstant(w3, *(uint32_t *)(OldCode + idx));
|
||||
cmp(w2, w3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
len -= 4;
|
||||
@@ -419,7 +286,7 @@ DEF_OP(ValidateCode) {
|
||||
while (len >= 2)
|
||||
{
|
||||
ldrh(w2, MemOperand(x0, idx));
|
||||
LoadConstant(w3, *(const uint16_t *)(OldCode + idx));
|
||||
LoadConstant(w3, *(uint16_t *)(OldCode + idx));
|
||||
cmp(w2, w3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
len -= 2;
|
||||
@@ -428,7 +295,7 @@ DEF_OP(ValidateCode) {
|
||||
while (len >= 1)
|
||||
{
|
||||
ldrb(w2, MemOperand(x0, idx));
|
||||
LoadConstant(w3, *(const uint8_t *)(OldCode + idx));
|
||||
LoadConstant(w3, *(uint8_t *)(OldCode + idx));
|
||||
cmp(w2, w3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
len -= 1;
|
||||
@@ -436,7 +303,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
DEF_OP(RemoveCodeEntry) {
|
||||
// Arguments are passed as follows:
|
||||
// X0: Thread
|
||||
// X1: RIP
|
||||
@@ -444,9 +311,9 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(x0, STATE);
|
||||
LoadConstant(x1, Entry);
|
||||
LoadConstant(x1, IR->GetHeader()->Entry);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)));
|
||||
LoadConstant(x2, reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit));
|
||||
SpillStaticRegs();
|
||||
blr(x2);
|
||||
FillStaticRegs();
|
||||
@@ -463,10 +330,19 @@ DEF_OP(CPUID) {
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
// x2 = CPUID Leaf
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)));
|
||||
mov(x1, GetReg<RA_64>(Op->Function.ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Leaf.ID()));
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&CTX->CPUID));
|
||||
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
|
||||
using ClassPtrType = FEXCore::CPUID::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t, uint32_t);
|
||||
union PtrCast {
|
||||
ClassPtrType ClassPtr;
|
||||
uintptr_t Data;
|
||||
};
|
||||
|
||||
PtrCast Ptr;
|
||||
Ptr.ClassPtr = &FEXCore::CPUIDEmu::RunFunction;
|
||||
LoadConstant(x3, Ptr.Data);
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
@@ -483,16 +359,18 @@ DEF_OP(CPUID) {
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(GUESTRETURN, GuestReturn);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
REGISTER_OP(JUMP, Jump);
|
||||
REGISTER_OP(CONDJUMP, CondJump);
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
@@ -10,28 +10,28 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
mov(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
mov(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
ins(GetDst(Node).V16B(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
ins(GetDst(Node).V8H(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
ins(GetDst(Node).V4S(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
ins(GetDst(Node).V2D(), Op->Index, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default: LogMan::Msg::A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,41 +39,45 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()).W());
|
||||
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
break;
|
||||
case 8:
|
||||
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()).X());
|
||||
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()).X());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
|
||||
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_U) {
|
||||
LogMan::Msg::A("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Src.ID()));
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Src.ID()));
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -81,17 +85,30 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvt(GetDst(Node).D(), GetSrc(Op->Scalar.ID()).S());
|
||||
fcvt(GetDst(Node).D(), GetSrc(Op->Header.Args[0].ID()).S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvt(GetDst(Node).S(), GetSrc(Op->Scalar.ID()).D());
|
||||
fcvt(GetDst(Node).S(), GetSrc(Op->Header.Args[0].ID()).D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
default: LogMan::Msg::A("Unknown FCVT sizes: 0x%x", Conv);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_UToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_UToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
ucvtf(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
ucvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -99,12 +116,25 @@ DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZU) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZU>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzu(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzu(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -112,12 +142,27 @@ DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToU) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToU>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
fcvtzu(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
fcvtzu(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -125,14 +170,14 @@ DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -142,70 +187,14 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2S());
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Vector.ID()).V2D());
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LogMan::Msg::A("Unknown Conversion Type : 0%04x", Conv); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -214,13 +203,16 @@ void Arm64JITCore::RegisterConversionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_U, Float_FromGPR_U);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_UTOF, Vector_UToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
REGISTER_OP(VECTOR_FTOZU, Vector_FToZU);
|
||||
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
|
||||
REGISTER_OP(VECTOR_FTOU, Vector_FToU);
|
||||
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
|
||||
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
|
||||
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,60 +10,61 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetDst(Node).V16B(), GetSrc(Op->Vector.ID()).V16B());
|
||||
aesimc(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesmc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesimc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
|
||||
aarch64::Literal ConstantLiteral (0x0C030609'0306090CULL, 0x040B0E01'0B0E0104ULL);
|
||||
aarch64::Label Constant;
|
||||
aarch64::Label PastConstant;
|
||||
|
||||
// Do a "regular" AESE step
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Src.ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
|
||||
// Do a table shuffle to undo ShiftRows
|
||||
ldr(VTMP3, &ConstantLiteral);
|
||||
adr(TMP1.X(), &Constant);
|
||||
ldr(VTMP3, MemOperand(TMP1.X()));
|
||||
|
||||
// Now EOR in the RCON
|
||||
if (Op->RCON) {
|
||||
@@ -79,68 +80,24 @@ DEF_OP(AESKeyGenAssist) {
|
||||
}
|
||||
|
||||
b(&PastConstant);
|
||||
place(&ConstantLiteral);
|
||||
bind(&Constant);
|
||||
dc32(0x0B0E0104);
|
||||
dc32(0x040B0E01);
|
||||
dc32(0x0306090C);
|
||||
dc32(0x0C030609);
|
||||
bind(&PastConstant);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
crc32cb(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 2:
|
||||
crc32ch(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 4:
|
||||
crc32cw(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 8:
|
||||
crc32cx(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_64>(Op->Src2.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
auto Dst = GetDst(Node).Q();
|
||||
auto Src1 = GetSrc(Op->Src1.ID()).V2D();
|
||||
auto Src2 = GetSrc(Op->Src2.ID()).V2D();
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
pmull(Dst, Src1, Src2);
|
||||
break;
|
||||
case 0b00000001:
|
||||
mov(VTMP1.V1D(), Src1, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src2);
|
||||
break;
|
||||
case 0b00010000:
|
||||
mov(VTMP1.V1D(), Src2, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src1);
|
||||
break;
|
||||
case 0b00010001:
|
||||
pmull2(Dst, Src1, Src2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -10,10 +10,10 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()), Op->Flag, 1);
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), Op->Flag, 1);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+366
-332
@@ -11,7 +11,6 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
@@ -21,14 +20,8 @@ $end_info$
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <sys/mman.h>
|
||||
@@ -36,55 +29,21 @@ $end_info$
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
namespace {
|
||||
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source / Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static int64_t LDIV(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source / Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static uint64_t LUREM(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source % Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static int64_t LREM(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source % Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void Arm64JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Original);
|
||||
ThreadSharedData = Core->ThreadSharedData;
|
||||
}
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Op: {}", FEXCore::IR::GetName(IROp->Op));
|
||||
#endif
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LogMan::Msg::A("Unhandled IR Op: %s", std::string(Name).c_str());
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16:{
|
||||
@@ -92,8 +51,9 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
LoadConstant(x1, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -108,7 +68,8 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
fmov(v0.S(), GetSrc(IROp->Args[0].ID()).S()) ;
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
LoadConstant(x0, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -127,7 +88,8 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
LoadConstant(x0, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -146,13 +108,9 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
if (Info.ABI == FABI_F80_I16) {
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
else {
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
LoadConstant(x1, (uintptr_t)Info.fn);
|
||||
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -171,9 +129,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -190,9 +149,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -203,52 +163,16 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
mov(v1.D(), GetSrc(IROp->Args[1].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
mov(GetDst(Node).D(), v0.D());
|
||||
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -264,9 +188,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -282,9 +207,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -300,12 +226,13 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
umov(x3, GetSrc(IROp->Args[1].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x4, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -321,9 +248,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x2, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -341,12 +269,13 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
|
||||
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
umov(x3, GetSrc(IROp->Args[1].ID()).V2D(), 1);
|
||||
|
||||
LoadConstant(x4, (uintptr_t)Info.fn);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -361,75 +290,141 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}",
|
||||
FEXCore::IR::GetName(IROp->Op), ToUnderlying(Info.ABI));
|
||||
#endif
|
||||
break;
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LogMan::Msg::A("Unhandled IR Fallback abi: %s %d", std::string(Name).c_str(), Info.ABI);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
emit.place(&l_BranchHost);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
});
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
void Arm64JITCore::Op_NoOp(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
Arm64JITCore::CodeBuffer Arm64JITCore::AllocateNewCodeBuffer(size_t Size) {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(
|
||||
mmap(nullptr,
|
||||
Buffer.Size,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS,
|
||||
-1, 0));
|
||||
LogMan::Throw::A(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
Dispatcher->RegisterCodeBuffer(Buffer.Ptr, Buffer.Size);
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx, 0)
|
||||
, HostSupportsSVE{ctx->HostFeatures.SupportsAVX}
|
||||
, CTX {ctx} {
|
||||
void Arm64JITCore::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
munmap(Buffer.Ptr, Buffer.Size);
|
||||
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
|
||||
}
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
bool Arm64JITCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
|
||||
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(ucontext);
|
||||
uint32_t Instr = PC[0];
|
||||
|
||||
if (!Dispatcher->IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
// 1 = 16bit
|
||||
// 2 = 32bit
|
||||
// 3 = 64bit
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
|
||||
0b1011'0000'0000; // Inner shareable all
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
uint32_t LDR = 0b0011'1000'0111'1111'0110'1000'0000'0000;
|
||||
LDR |= Size << 30;
|
||||
LDR |= AddrReg << 5;
|
||||
LDR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
uint32_t STR = 0b0011'1000'0011'1111'0110'1000'0000'0000;
|
||||
STR |= Size << 30;
|
||||
STR |= AddrReg << 5;
|
||||
STR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STR;
|
||||
PC[1] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS CASPAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS CASAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x: PC: %p Instruction: 0x%08x\n", Op, PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Unhandled JIT SIGBUS: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
|
||||
return false;
|
||||
}
|
||||
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(&PC[-1], 16);
|
||||
return true;
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
: Arm64Emitter(0)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread} {
|
||||
{
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
config.StaticRegisterAssignment = true;
|
||||
|
||||
Dispatcher = new Arm64Dispatcher(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
}
|
||||
|
||||
// Can't allocate a code buffer until after dispatcher is created
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
|
||||
*GetBuffer() = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
SetAllowAssembler(true);
|
||||
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
|
||||
RAPass = Thread->PassManager->GetRAPass();
|
||||
|
||||
#if DEBUG
|
||||
Decoder.AppendVisitor(&Disasm)
|
||||
@@ -468,157 +463,159 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
RegisterVectorHandlers();
|
||||
RegisterEncryptionHandlers();
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
if (!CompileThread) {
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalReturnInstruction = Dispatcher->SignalHandlerReturnAddress;
|
||||
|
||||
// Common
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
// This will register the host signal handler per thread, which is fine
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
|
||||
});
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->HandleSIGBUS(Signal, info, ucontext);
|
||||
});
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
});
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::Context::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
|
||||
|
||||
// Platform Specific
|
||||
auto &AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
AArch64.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
AArch64.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
SetAllowAssembler(true);
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Thread->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
const char JITString[] = "FEXJIT::Arm64JITCore::";
|
||||
auto Buffer = GetBuffer();
|
||||
Buffer->EmitString(JITString);
|
||||
Buffer->Align();
|
||||
}
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
*GetBuffer() = vixl::CodeBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
auto Buffer = GetBuffer();
|
||||
if (*ThreadSharedData.SignalHandlerRefCounterPtr == 0) {
|
||||
if (!CodeBuffers.empty()) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
// Set the current code buffer to the initial
|
||||
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
}
|
||||
|
||||
if (CurrentCodeBuffer->Size == MAX_CODE_SIZE) {
|
||||
// Rewind to the start of the code cache start
|
||||
Buffer->Reset();
|
||||
}
|
||||
else {
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
InitialCodeBuffer.Size *= 1.5;
|
||||
InitialCodeBuffer.Size = std::min(InitialCodeBuffer.Size, MAX_CODE_SIZE);
|
||||
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(InitialCodeBuffer.Size);
|
||||
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = Arm64JITCore::AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
*Buffer = vixl::CodeBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
}
|
||||
|
||||
IR::PhysicalRegister Arm64JITCore::GetPhys(IR::NodeID Node) const {
|
||||
static IR::PhysicalRegister GetPhys(IR::RegisterAllocationData *RAData, uint32_t Node) {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
LogMan::Throw::A(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa%d. Class: %d", Node, PhyReg.Class);
|
||||
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(uint32_t Node) {
|
||||
auto Reg = GetPhys(RAData, Node);
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg].W();
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg].W();
|
||||
} else {
|
||||
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(uint32_t Node) {
|
||||
auto Reg = GetPhys(RAData, Node);
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
} else {
|
||||
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_32>(IR::NodeID Node) const {
|
||||
uint32_t Reg = GetPhys(Node).Reg;
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_32>(uint32_t Node) {
|
||||
uint32_t Reg = GetPhys(RAData, Node).Reg;
|
||||
return RA32Pair[Reg];
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_64>(IR::NodeID Node) const {
|
||||
uint32_t Reg = GetPhys(Node).Reg;
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_64>(uint32_t Node) {
|
||||
uint32_t Reg = GetPhys(RAData, Node).Reg;
|
||||
return RA64Pair[Reg];
|
||||
}
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(uint32_t Node) {
|
||||
auto Reg = GetPhys(RAData, Node);
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetDst(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetDst(uint32_t Node) {
|
||||
auto Reg = GetPhys(RAData, Node);
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINECONSTANT) {
|
||||
@@ -632,18 +629,13 @@ bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_
|
||||
}
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = OpHeader->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
*Value = (Entry + Op->Offset) & Mask;
|
||||
*Value = IR->GetHeader()->Entry + Op->Offset;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
@@ -651,46 +643,42 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(IR::NodeID Node) const {
|
||||
return FEXCore::IR::RegisterClassType {GetPhys(Node).Class};
|
||||
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(uint32_t Node) {
|
||||
return FEXCore::IR::RegisterClassType {GetPhys(RAData, Node).Class};
|
||||
}
|
||||
|
||||
|
||||
bool Arm64JITCore::IsFPR(IR::NodeID Node) const {
|
||||
bool Arm64JITCore::IsFPR(uint32_t Node) {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
bool Arm64JITCore::IsGPR(uint32_t Node) {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
void *Arm64JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
using namespace aarch64;
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
#ifndef NDEBUG
|
||||
LoadConstant(x0, Entry);
|
||||
LoadConstant(x0, HeaderOp->Entry);
|
||||
#endif
|
||||
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState, false);
|
||||
}
|
||||
|
||||
// AAPCS64
|
||||
@@ -713,14 +701,35 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
// X1-X3 = Temp
|
||||
// X4-r18 = RA
|
||||
|
||||
GuestEntry = GetCursorAddress<uint8_t *>();
|
||||
auto Buffer = GetBuffer();
|
||||
auto Entry = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
GetBuffer()->CursorForward(GDBSize);
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
aarch64::Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Thread))); // Get thread
|
||||
ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
|
||||
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
cbz(w0, &RunBlock);
|
||||
{
|
||||
// Make sure RIP is syncronized to the context
|
||||
LoadConstant(x0, HeaderOp->Entry);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
|
||||
// Stop the thread
|
||||
LoadConstant(x0, Dispatcher->ThreadPauseHandlerAddressSpillSRA);
|
||||
br(x0);
|
||||
}
|
||||
bind(&RunBlock);
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
//LogMan::Throw::A(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
@@ -737,15 +746,15 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t *>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
uint32_t Node = IR->GetID(BlockNode);
|
||||
auto IsTarget = JumpTargets.find(Node);
|
||||
if (IsTarget == JumpTargets.end()) {
|
||||
IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
}
|
||||
|
||||
// if there's a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
|
||||
@@ -757,8 +766,12 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
bind(&IsTarget->second);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()), 0, IR->GetID(BlockNode)});
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
uint32_t ID = IR->GetID(CodeNode);
|
||||
|
||||
// Execute handler
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
@@ -766,10 +779,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
|
||||
static_cast<uint32_t>(GetCursorAddress<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
DebugData->Subblocks.back().HostCodeSize = Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()) - DebugData->Subblocks.back().HostCodeStart;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -782,44 +792,68 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
|
||||
FinalizeCode();
|
||||
|
||||
auto CodeEnd = GetCursorAddress<uint8_t *>();
|
||||
CPU.EnsureIAndDCacheCoherency(GuestEntry, CodeEnd - GuestEntry);
|
||||
auto CodeEnd = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
|
||||
CPU.EnsureIAndDCacheCoherency(reinterpret_cast<void*>(Entry), CodeEnd - reinterpret_cast<uint64_t>(Entry));
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = CodeEnd - GuestEntry;
|
||||
DebugData->Relocations = &Relocations;
|
||||
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(CodeEnd) - reinterpret_cast<uintptr_t>(Entry);
|
||||
}
|
||||
|
||||
this->IR = nullptr;
|
||||
|
||||
return GuestEntry;
|
||||
return reinterpret_cast<void*>(Entry);
|
||||
}
|
||||
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
return;
|
||||
uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//printf("ExitFunctionLink: Aborting, %lX not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return core->Dispatcher->AbsoluteLoopTopAddress;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = core->Dispatcher->ExitFunctionLinkerAddress;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
emit.place(&l_BranchHost);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
});
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
FEXCore::CPU::CPUBackend *CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return new Arm64JITCore(ctx, Thread, CompileThread);
|
||||
}
|
||||
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
Arm64JITCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsStaticRegisterAllocation = true
|
||||
};
|
||||
}
|
||||
|
||||
}
|
||||
+87
-142
@@ -6,23 +6,18 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#define STATE x28
|
||||
#define TMP1 x0
|
||||
#define TMP2 x1
|
||||
@@ -43,37 +38,38 @@ using namespace vixl::aarch64;
|
||||
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
public:
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
struct CodeBuffer {
|
||||
uint8_t *Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
|
||||
~Arm64JITCore() override;
|
||||
std::string GetName() override { return "JIT"; }
|
||||
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
bool HandleSIGBUS(int Signal, void *info, void *ucontext);
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
|
||||
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
const bool HostSupportsSVE{};
|
||||
|
||||
Dispatcher *Dispatcher;
|
||||
Label *PendingTargetLabel;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
uint64_t Entry;
|
||||
|
||||
std::map<IR::NodeID, aarch64::Label> JumpTargets;
|
||||
std::map<IR::OrderedNodeWrapper::NodeOffsetType, aarch64::Label> JumpTargets;
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
@@ -97,39 +93,33 @@ private:
|
||||
constexpr static uint8_t RA_FPR = 2;
|
||||
|
||||
template<uint8_t RAType>
|
||||
[[nodiscard]] aarch64::Register GetReg(IR::NodeID Node) const;
|
||||
aarch64::Register GetReg(uint32_t Node);
|
||||
|
||||
template<>
|
||||
[[nodiscard]] aarch64::Register GetReg<RA_32>(IR::NodeID Node) const;
|
||||
aarch64::Register GetReg<RA_32>(uint32_t Node);
|
||||
template<>
|
||||
[[nodiscard]] aarch64::Register GetReg<RA_64>(IR::NodeID Node) const;
|
||||
aarch64::Register GetReg<RA_64>(uint32_t Node);
|
||||
|
||||
template<uint8_t RAType>
|
||||
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair(IR::NodeID Node) const;
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair(uint32_t Node);
|
||||
|
||||
template<>
|
||||
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_32>(IR::NodeID Node) const;
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_32>(uint32_t Node);
|
||||
template<>
|
||||
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_64>(IR::NodeID Node) const;
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_64>(uint32_t Node);
|
||||
|
||||
[[nodiscard]] aarch64::VRegister GetSrc(IR::NodeID Node) const;
|
||||
[[nodiscard]] aarch64::VRegister GetDst(IR::NodeID Node) const;
|
||||
aarch64::VRegister GetSrc(uint32_t Node);
|
||||
aarch64::VRegister GetDst(uint32_t Node);
|
||||
|
||||
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
FEXCore::IR::RegisterClassType GetRegClass(uint32_t Node);
|
||||
|
||||
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const;
|
||||
bool IsFPR(uint32_t Node);
|
||||
bool IsGPR(uint32_t Node);
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
MemOperand GenerateMemOperand(uint8_t AccessSize, aarch64::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]] MemOperand GenerateMemOperand(uint8_t AccessSize,
|
||||
aarch64::Register Base,
|
||||
IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr);
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value);
|
||||
|
||||
struct LiveRange {
|
||||
uint32_t Begin;
|
||||
@@ -140,80 +130,46 @@ private:
|
||||
vixl::aarch64::Decoder Decoder;
|
||||
#endif
|
||||
|
||||
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
|
||||
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
|
||||
}
|
||||
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
// This is the initial code buffer that we will fall back to
|
||||
// In a program without signals and code clearing, we will typically
|
||||
// only have this code buffer
|
||||
CodeBuffer InitialCodeBuffer{};
|
||||
// This is the array of /additional/ code buffers that we may need to allocate
|
||||
// Allocation only occurs when we've hit signals and need to clear code cache
|
||||
// For code safety we can't delete code buffers until outside of all signals
|
||||
std::vector<CodeBuffer> CodeBuffers{};
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer *CurrentCodeBuffer{};
|
||||
|
||||
// We don't want to mvoe above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
|
||||
|
||||
#if DEBUG
|
||||
vixl::aarch64::Disassembler Disasm;
|
||||
#endif
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
static uint64_t ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalReturnInstruction{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
};
|
||||
|
||||
CompilerSharedData ThreadSharedData;
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
FEXCore::Core::DebugData *DebugData;
|
||||
|
||||
void ResetStack();
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Literal<uint64_t> Lit;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
using OpHandler = void (Arm64JITCore::*)(FEXCore::IR::IROp_Header *IROp, uint32_t Node);
|
||||
std::array<OpHandler, FEXCore::IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
void RegisterALUHandlers();
|
||||
void RegisterAtomicHandlers();
|
||||
void RegisterBranchHandlers();
|
||||
@@ -224,7 +180,7 @@ private:
|
||||
void RegisterMoveHandlers();
|
||||
void RegisterVectorHandlers();
|
||||
void RegisterEncryptionHandlers();
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
@@ -252,7 +208,6 @@ private:
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
DEF_OP(Lshl);
|
||||
DEF_OP(Lshr);
|
||||
@@ -260,8 +215,6 @@ private:
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
@@ -281,6 +234,7 @@ private:
|
||||
DEF_OP(VExtractToGPR);
|
||||
DEF_OP(Float_ToGPR_ZU);
|
||||
DEF_OP(Float_ToGPR_ZS);
|
||||
DEF_OP(Float_ToGPR_U);
|
||||
DEF_OP(Float_ToGPR_S);
|
||||
DEF_OP(FCmp);
|
||||
|
||||
@@ -298,37 +252,41 @@ private:
|
||||
DEF_OP(AtomicFetchAnd);
|
||||
DEF_OP(AtomicFetchOr);
|
||||
DEF_OP(AtomicFetchXor);
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(GuestReturn);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
DEF_OP(Jump);
|
||||
DEF_OP(CondJump);
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(RemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(Float_FromGPR_U);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_UToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
DEF_OP(Vector_FToZU);
|
||||
DEF_OP(Vector_FToZS);
|
||||
DEF_OP(Vector_FToU);
|
||||
DEF_OP(Vector_FToS);
|
||||
DEF_OP(Vector_FToF);
|
||||
DEF_OP(Vector_FToI);
|
||||
|
||||
///< Flag ops
|
||||
DEF_OP(GetHostFlag);
|
||||
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
@@ -342,15 +300,11 @@ private:
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(LoadMemTSO);
|
||||
DEF_OP(StoreMemTSO);
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(EndBlock);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Phi);
|
||||
@@ -358,9 +312,6 @@ private:
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -370,11 +321,12 @@ private:
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(CreateVector2);
|
||||
DEF_OP(CreateVector4);
|
||||
DEF_OP(SplatVector2);
|
||||
DEF_OP(SplatVector4);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
DEF_OP(VOr);
|
||||
DEF_OP(VXor);
|
||||
DEF_OP(VAdd);
|
||||
@@ -385,10 +337,8 @@ private:
|
||||
DEF_OP(VSQSub);
|
||||
DEF_OP(VAddP);
|
||||
DEF_OP(VAddV);
|
||||
DEF_OP(VUMinV);
|
||||
DEF_OP(VURAvg);
|
||||
DEF_OP(VAbs);
|
||||
DEF_OP(VPopcount);
|
||||
DEF_OP(VFAdd);
|
||||
DEF_OP(VFAddP);
|
||||
DEF_OP(VFSub);
|
||||
@@ -408,8 +358,6 @@ private:
|
||||
DEF_OP(VSMax);
|
||||
DEF_OP(VZip);
|
||||
DEF_OP(VZip2);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VUnZip2);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
@@ -432,7 +380,6 @@ private:
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
@@ -455,9 +402,7 @@ private:
|
||||
DEF_OP(VSMull);
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
@@ -466,9 +411,9 @@ private:
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
|
||||
}
|
||||
|
||||
+123
-368
@@ -4,16 +4,13 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
@@ -32,7 +29,7 @@ DEF_OP(LoadContext) {
|
||||
case 8:
|
||||
ldr(GetReg<RA_64>(Node), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unhandled LoadContext size: %d", OpSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -53,33 +50,33 @@ DEF_OP(LoadContext) {
|
||||
case 16:
|
||||
ldr(Dst, MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unhandled LoadContext size: %d", OpSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
strb(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
strb(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 2:
|
||||
strh(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
strh(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 4:
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
str(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 8:
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unhandled StoreContext size: %d", OpSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
auto Src = GetSrc(Op->Header.Args[0].ID());
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
str(Src.B(), MemOperand(STATE, Op->Offset));
|
||||
@@ -96,7 +93,7 @@ DEF_OP(StoreContext) {
|
||||
case 16:
|
||||
str(Src, MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unhandled LoadContext size: %d", OpSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -106,60 +103,58 @@ DEF_OP(LoadRegister) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0])) / 8;
|
||||
auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LogMan::Throw::A(regId < SRA64.size(), "out of range regId");
|
||||
|
||||
auto reg = SRA64[regId];
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
ubfx(GetReg<RA_64>(Node), reg, regOffs * 8, 8);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
ubfx(GetReg<RA_64>(Node), reg, 0, 16);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
|
||||
mov(GetReg<RA_32>(Node), reg.W());
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
|
||||
mov(GetReg<RA_64>(Node), reg);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = CTX->HostFeatures.SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
|
||||
auto regOffs = Op->Offset & 15;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
LogMan::Throw::A(regId < SRAFPR.size(), "out of range regId");
|
||||
|
||||
auto guest = SRAFPR[regId];
|
||||
auto host = GetSrc(Node);
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
mov(host.B(), guest.B());
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
fmov(host.H(), guest.H());
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A((regOffs & 3) == 0, "unexpected regOffs");
|
||||
if (regOffs == 0) {
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
fmov(host.S(), guest.S());
|
||||
@@ -169,7 +164,7 @@ DEF_OP(LoadRegister) {
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A((regOffs & 7) == 0, "unexpected regOffs");
|
||||
if (regOffs == 0) {
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
mov(host.D(), guest.D());
|
||||
@@ -179,13 +174,13 @@ DEF_OP(LoadRegister) {
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
mov(host.Q(), guest.Q());
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LogMan::Throw::A(false, "Unhandled Op->Class %d", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -193,42 +188,40 @@ DEF_OP(StoreRegister) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
auto regId = Op->Offset / 8 - 1;
|
||||
auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LogMan::Throw::A(regId < SRA64.size(), "out of range regId");
|
||||
|
||||
auto reg = SRA64[regId];
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), regOffs * 8, 8);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 16);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 32);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Op->Value.ID()).GetCode() != reg.GetCode())
|
||||
mov(reg, GetReg<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = CTX->HostFeatures.SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
|
||||
auto regOffs = Op->Offset & 15;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
LogMan::Throw::A(regId < SRAFPR.size(), "regId out of range");
|
||||
|
||||
auto guest = SRAFPR[regId];
|
||||
auto host = GetSrc(Op->Value.ID());
|
||||
@@ -239,36 +232,36 @@ DEF_OP(StoreRegister) {
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A((regOffs & 1) == 0, "unexpected regOffs");
|
||||
ins(guest.V8H(), regOffs/2, host.V8H(), 0);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A((regOffs & 3) == 0, "unexpected regOffs");
|
||||
ins(guest.V4S(), regOffs/4, host.V4S(), 0);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A((regOffs & 7) == 0, "unexpected regOffs");
|
||||
ins(guest.V2D(), regOffs / 8, host.V2D(), 0);
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
|
||||
if (guest.GetCode() != host.GetCode())
|
||||
mov(guest.Q(), host.Q());
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LogMan::Throw::A(false, "Unhandled Op->Class %d", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Index.ID());
|
||||
size_t size = Op->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -294,17 +287,15 @@ DEF_OP(LoadContextIndexed) {
|
||||
ldr(GetReg<RA_64>(Node), MemOperand(TMP1, Op->BaseOffset));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16:
|
||||
LOGMAN_MSG_A_FMT("Invalid Class load of size 16");
|
||||
LogMan::Msg::A("Invalid Class load of size 16");
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed stride: {}", Op->Stride);
|
||||
break;
|
||||
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -341,25 +332,23 @@ DEF_OP(LoadContextIndexed) {
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed stride: {}", Op->Stride);
|
||||
break;
|
||||
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Index.ID());
|
||||
size_t size = Op->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto value = GetReg<RA_64>(Op->Value.ID());
|
||||
auto value = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -384,21 +373,19 @@ DEF_OP(StoreContextIndexed) {
|
||||
str(value, MemOperand(TMP1, Op->BaseOffset));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16:
|
||||
LOGMAN_MSG_A_FMT("Invalid Class store of size 16");
|
||||
LogMan::Msg::A("Invalid Class load of size 16");
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed stride: {}", Op->Stride);
|
||||
break;
|
||||
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto value = GetSrc(Op->Value.ID());
|
||||
auto value = GetSrc(Op->Header.Args[0].ID());
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -433,68 +420,66 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
|
||||
break;
|
||||
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed stride: {}", Op->Stride);
|
||||
break;
|
||||
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * 16;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint32_t SlotOffset = Op->Slot * 16 + 16;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
strb(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
strh(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
strh(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
str(GetSrc(Op->Value.ID()).S(), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Header.Args[0].ID()).S(), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
str(GetSrc(Op->Value.ID()).D(), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Header.Args[0].ID()).D(), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
str(GetSrc(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
LogMan::Msg::A("Unhandled SpillRegister class: %d", Op->Class.Val);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(FillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_FillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
uint32_t SlotOffset = Op->Slot * 16 + 16;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
@@ -514,7 +499,7 @@ DEF_OP(FillRegister) {
|
||||
ldr(GetReg<RA_64>(Node), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
switch (OpSize) {
|
||||
@@ -530,10 +515,10 @@ DEF_OP(FillRegister) {
|
||||
ldr(GetDst(Node), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
LogMan::Msg::A("Unhandled FillRegister class: %d", Op->Class.Val);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -545,7 +530,7 @@ DEF_OP(LoadFlag) {
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemOperand(STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag));
|
||||
strb(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag));
|
||||
}
|
||||
|
||||
MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
@@ -553,7 +538,7 @@ MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Registe
|
||||
return MemOperand(Base);
|
||||
} else {
|
||||
if (OffsetScale != 1 && OffsetScale != AccessSize) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetScale: {}", OffsetScale);
|
||||
LogMan::Msg::A("Unhandled GenerateMemOperand OffsetScale: %d", OffsetScale);
|
||||
}
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Offset, &Const)) {
|
||||
@@ -565,23 +550,22 @@ MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Registe
|
||||
case IR::MEM_OFFSET_UXTW.Val: return MemOperand(Base, RegOffset.W(), Extend::UXTW, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTW.Val: return MemOperand(Base, RegOffset.W(), Extend::SXTW, (int)std::log2(OffsetScale) );
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
|
||||
default: LogMan::Msg::A("Unhandled GenerateMemOperand OffsetType: %d", OffsetType.Val); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemSrc = GenerateMemOperand(Op->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1:
|
||||
ldrb(Dst, MemSrc);
|
||||
break;
|
||||
@@ -594,12 +578,12 @@ DEF_OP(LoadMem) {
|
||||
case 8:
|
||||
ldr(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Dst = GetDst(Node);
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1:
|
||||
ldr(Dst.B(), MemSrc);
|
||||
break;
|
||||
@@ -615,7 +599,7 @@ DEF_OP(LoadMem) {
|
||||
case 16:
|
||||
ldr(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -623,44 +607,14 @@ DEF_OP(LoadMem) {
|
||||
DEF_OP(LoadMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9) {
|
||||
// RCPC2 means that the offset must be an inline constant
|
||||
LOGMAN_THROW_A_FMT(MemSrc.IsRegisterOffset() == false, "RCPC2 doesn't support register offset. Only Immediate offset");
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "LoadMemTSO: No offset allowed");
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LogMan::Msg::A("LoadMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldapurb(Dst, MemSrc);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
ldapur(Dst.W(), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
ldapur(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
if (SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldaprb(Dst, MemSrc);
|
||||
@@ -669,7 +623,7 @@ DEF_OP(LoadMemTSO) {
|
||||
// Aligned
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
nop();
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 2:
|
||||
ldaprh(Dst, MemSrc);
|
||||
break;
|
||||
@@ -679,13 +633,13 @@ DEF_OP(LoadMemTSO) {
|
||||
case 8:
|
||||
ldapr(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
if (Op->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldarb(Dst, MemSrc);
|
||||
@@ -694,7 +648,7 @@ DEF_OP(LoadMemTSO) {
|
||||
// Aligned
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
nop();
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 2:
|
||||
ldarh(Dst, MemSrc);
|
||||
break;
|
||||
@@ -704,7 +658,7 @@ DEF_OP(LoadMemTSO) {
|
||||
case 8:
|
||||
ldar(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
@@ -712,7 +666,7 @@ DEF_OP(LoadMemTSO) {
|
||||
else {
|
||||
dmb(InnerShareable, BarrierAll);
|
||||
auto Dst = GetDst(Node);
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 2:
|
||||
ldr(Dst.H(), MemSrc);
|
||||
break;
|
||||
@@ -725,7 +679,7 @@ DEF_OP(LoadMemTSO) {
|
||||
case 16:
|
||||
ldr(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
|
||||
}
|
||||
dmb(InnerShareable, BarrierAll);
|
||||
}
|
||||
@@ -734,30 +688,30 @@ DEF_OP(LoadMemTSO) {
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
auto MemSrc = GenerateMemOperand(Op->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 1:
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
strb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
case 2:
|
||||
strh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
strh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
str(GetReg<RA_32>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
str(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
switch (IROp->Size) {
|
||||
auto Src = GetSrc(Op->Header.Args[1].ID());
|
||||
switch (Op->Size) {
|
||||
case 1:
|
||||
str(Src.B(), MemSrc);
|
||||
break;
|
||||
@@ -773,73 +727,45 @@ DEF_OP(StoreMem) {
|
||||
case 16:
|
||||
str(Src, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9) {
|
||||
// RCPC2 means that the offset must be an inline constant
|
||||
LOGMAN_THROW_A_FMT(MemSrc.IsRegisterOffset() == false, "RCPC2 doesn't support register offset. Only Immediate offset");
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "StoreMemTSO: No offset allowed");
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LogMan::Msg::A("StoreMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
stlrb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
nop();
|
||||
switch (IROp->Size) {
|
||||
switch (Op->Size) {
|
||||
case 2:
|
||||
stlurh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
stlrh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
stlur(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
stlr(GetReg<RA_32>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
stlur(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
stlr(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
nop();
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
stlrh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
stlr(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
stlr(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else {
|
||||
dmb(InnerShareable, BarrierAll);
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
switch (IROp->Size) {
|
||||
auto Src = GetSrc(Op->Header.Args[1].ID());
|
||||
switch (Op->Size) {
|
||||
case 1:
|
||||
str(Src.B(), MemSrc);
|
||||
break;
|
||||
@@ -855,181 +781,18 @@ DEF_OP(StoreMemTSO) {
|
||||
case 16:
|
||||
str(Src, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
|
||||
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
|
||||
}
|
||||
dmb(InnerShareable, BarrierAll);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Addr.ID()));
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidLoadMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
ldarb(Dst, MemSrc);
|
||||
}
|
||||
else {
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldarh(Dst, MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
ldar(Dst.W(), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
ldar(Dst, MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Dst = GetDst(Node);
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
ldarh(TMP1.W(), MemSrc);
|
||||
fmov(Dst.H(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
ldar(TMP1.W(), MemSrc);
|
||||
fmov(Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 8:
|
||||
ldar(TMP1, MemSrc);
|
||||
fmov(Dst.D(), TMP1);
|
||||
break;
|
||||
case 16:
|
||||
nop();
|
||||
ldaxp(TMP1, TMP2, MemSrc);
|
||||
clrex();
|
||||
mov(Dst.V2D(), 0, TMP1);
|
||||
mov(Dst.V2D(), 1, TMP2);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidStoreMemTSO) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Addr.ID()));
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidStoreMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
}
|
||||
else {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
stlrh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
stlr(GetReg<RA_32>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
stlr(GetReg<RA_64>(Op->Value.ID()), MemSrc);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
mov(TMP1.W(), Src.V16B(), 0);
|
||||
stlrb(TMP1, MemSrc);
|
||||
}
|
||||
else {
|
||||
switch (IROp->Size) {
|
||||
case 2:
|
||||
mov(TMP1.W(), Src.V8H(), 0);
|
||||
stlrh(TMP1, MemSrc);
|
||||
break;
|
||||
case 4:
|
||||
mov(TMP1.W(), Src.V4S(), 0);
|
||||
stlr(TMP1.W(), MemSrc);
|
||||
break;
|
||||
case 8:
|
||||
mov(TMP1, Src.V2D(), 0);
|
||||
stlr(TMP1, MemSrc);
|
||||
break;
|
||||
case 16: {
|
||||
// Move vector to GPRs
|
||||
mov(TMP1, Src.V2D(), 0);
|
||||
mov(TMP2, Src.V2D(), 1);
|
||||
Label B;
|
||||
bind(&B);
|
||||
|
||||
// ldaxp must not have both the destination registers be the same
|
||||
ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(TMP3, TMP1, TMP2, MemSrc); // <- Can also hit SIGBUS
|
||||
cbnz(TMP3, &B); // < Overwritten with DMB
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
LogMan::Msg::A("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(VStoreMemElement) {
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
mov(TMP1, MemReg);
|
||||
for (size_t i = 0; i < std::max(1U, CTX->HostFeatures.DCacheLineSize / 64U); ++i) {
|
||||
dc(DataCacheOp::CVAU, TMP1);
|
||||
add(TMP1, TMP1, CTX->HostFeatures.DCacheLineSize);
|
||||
}
|
||||
dsb(InnerShareable, BarrierAll);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
// We can use this instruction directly
|
||||
dc(DataCacheOp::ZVA, MemReg);
|
||||
}
|
||||
else {
|
||||
// We must walk the cacheline ourselves
|
||||
// Force cacheline alignment
|
||||
and_(TMP1, MemReg, ~(CPUIDEmu::CACHELINE_SIZE - 1));
|
||||
// This will end up being four STPs
|
||||
// Depending on uarch it could be slightly more efficient in instructions emitted
|
||||
// and uops to use vector pair STP, but we want the non-temporal bit specifically here
|
||||
for (size_t i = 0; i < CPUIDEmu::CACHELINE_SIZE; i += 16) {
|
||||
stnp(xzr, xzr, MemOperand(TMP1, i, Offset));
|
||||
}
|
||||
}
|
||||
LogMan::Msg::A("Unimplemented");
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
@@ -1047,18 +810,10 @@ void Arm64JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(STOREFLAG, StoreFlag);
|
||||
REGISTER_OP(LOADMEM, LoadMem);
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
if (ParanoidTSO()) {
|
||||
REGISTER_OP(LOADMEMTSO, ParanoidLoadMemTSO);
|
||||
REGISTER_OP(STOREMEMTSO, ParanoidStoreMemTSO);
|
||||
}
|
||||
else {
|
||||
REGISTER_OP(LOADMEMTSO, LoadMemTSO);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMemTSO);
|
||||
}
|
||||
REGISTER_OP(LOADMEMTSO, LoadMemTSO);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMemTSO);
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+27
-143
@@ -4,20 +4,13 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include <syscall.h>
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, GetCursorAddress<uint8_t*>() - GuestEntry});
|
||||
}
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
@@ -31,44 +24,36 @@ DEF_OP(Fence) {
|
||||
case IR::Fence_Store.Val:
|
||||
dmb(FullSystem, BarrierWrites);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
default: LogMan::Msg::A("Unknown Fence: %d", Op->Fence); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case 0: // Hard fault
|
||||
case 5: // Guest ud2
|
||||
hlt(4);
|
||||
break;
|
||||
case 4: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
add(sp, TMP1, 0);
|
||||
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
// Now we need to jump to the thread stop handler
|
||||
LoadConstant(TMP1, Dispatcher->ThreadStopHandlerAddressSpillSRA);
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
case 6: { // INT3
|
||||
ResetStack();
|
||||
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, Op->Reason.Signal);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)));
|
||||
LoadConstant(w1, Op->Reason.TrapNumber);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)));
|
||||
LoadConstant(w1, Op->Reason.si_code);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)));
|
||||
LoadConstant(x1, Op->Reason.ErrorRegister);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV)));
|
||||
br(TMP1);
|
||||
break;
|
||||
default:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
br(TMP1);
|
||||
break;
|
||||
LoadConstant(TMP1, Dispatcher->ThreadPauseHandlerAddressSpillSRA);
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::A("Unknown Break reason: %d", Op->Reason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -100,7 +85,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetReg<RA_64>(Op->RoundMode.ID());
|
||||
auto Src = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
// Setup the rounding flags correctly
|
||||
and_(TMP1, Src, 0b11);
|
||||
@@ -130,102 +115,6 @@ DEF_OP(SetRoundingMode) {
|
||||
msr(FPCR, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Value.ID()));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue)));
|
||||
}
|
||||
else {
|
||||
fmov(x0, GetSrc(Op->Value.ID()).V1D());
|
||||
// Bug in vixl that source vector needs to b V1D rather than V2D?
|
||||
fmov(x1, GetSrc(Op->Value.ID()).V1D(), 1);
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)));
|
||||
}
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(x0, SpillMask & 0xFFFF);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
// Allocate some temporary space for storing the uint32_t CPU and Node IDs
|
||||
sub(sp, sp, 16);
|
||||
|
||||
// Load the getcpu syscall number
|
||||
LoadConstant(x8, SYS_getcpu);
|
||||
|
||||
// CPU pointer in x0
|
||||
add(x0, sp, 0);
|
||||
// Node in x1
|
||||
add(x1, sp, 4);
|
||||
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
// Load the values returned by the kernel
|
||||
ldp(w0, w1, MemOperand(sp));
|
||||
// Deallocate stack space
|
||||
sub(sp, sp, 16);
|
||||
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
|
||||
// Now store the result in the destination in the expected format
|
||||
// uint32_t Res = (node << 12) | cpu;
|
||||
// CPU is in w0
|
||||
// Node is in w1
|
||||
orr(GetReg<RA_64>(Node), x0, Operand(x1, LSL, 12));
|
||||
}
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
|
||||
if (Op->GetReseeded) {
|
||||
mrs(Dst.first, RNDRRS);
|
||||
}
|
||||
else {
|
||||
mrs(Dst.first, RNDR);
|
||||
}
|
||||
|
||||
// If the rng number is valid then NZCV is 0b0000, otherwise NZCV is 0b0100
|
||||
cset(Dst.second, Condition::ne);
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
hint(SystemHint::YIELD);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -234,19 +123,14 @@ void Arm64JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
REGISTER_OP(PHIVALUE, NoOp);
|
||||
REGISTER_OP(PRINT, Print);
|
||||
REGISTER_OP(PRINT, Unhandled);
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+11
-11
@@ -10,23 +10,23 @@ namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
auto Src = GetSrcPair<RA_32>(Op->Pair.ID());
|
||||
auto Src = GetSrcPair<RA_32>(Op->Header.Args[0].ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_32>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = GetSrcPair<RA_64>(Op->Pair.ID());
|
||||
auto Src = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_64>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Size"); break;
|
||||
default: LogMan::Msg::A("Unknown Size"); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -37,22 +37,22 @@ DEF_OP(CreateElementPair) {
|
||||
aarch64::Register RegSecond;
|
||||
aarch64::Register RegTmp;
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetReg<RA_32>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Upper.ID());
|
||||
RegFirst = GetReg<RA_32>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Header.Args[1].ID());
|
||||
RegTmp = w0;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetReg<RA_64>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Upper.ID());
|
||||
RegFirst = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
RegTmp = x0;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Size"); break;
|
||||
default: LogMan::Msg::A("Unknown Size"); break;
|
||||
}
|
||||
|
||||
if (Dst.first.GetCode() != RegSecond.GetCode()) {
|
||||
@@ -70,7 +70,7 @@ DEF_OP(CreateElementPair) {
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()));
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
Loaded 100 of 963 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user