Compare commits

..
747 changed files with 55578 additions and 120575 deletions

No files matched your search

+50 -50
View File
@@ -17,11 +17,11 @@ env:
FEX_ENABLEAVX: 1
jobs:
build_plus_test:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARMv8.0], [self-hosted, ARMv8.2], [self-hosted, ARMv8.4]]
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2], [self-hosted, ARMv8.4]]
fail-fast: false
steps:
@@ -65,7 +65,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -78,6 +78,18 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -90,6 +102,30 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: gvisor tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gvisor_tests
- name: GVisor Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GVisor.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -114,6 +150,17 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
- name: Struct verifier tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target struct_verifier
- name: Struct verifier Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
- name: APITest tests
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -197,53 +244,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkResults.log || true
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: gvisor tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gvisor_tests
- name: GVisor Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GVisor.log || true
- name: Struct verifier tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target struct_verifier
- name: Struct verifier Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
+28 -27
View File
@@ -24,11 +24,12 @@ env:
FEX_ENABLEAVX: 1
jobs:
glibc_fault_test:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARM64]]
# Run on an x86 device and any ARM runner.
arch: [[self-hosted, x64], [self-hosted, ARM64]]
fail-fast: false
steps:
@@ -72,7 +73,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -85,6 +86,18 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -97,6 +110,18 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -154,30 +179,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
-107
View File
@@ -1,107 +0,0 @@
name: Hostrunner tests
on:
push:
branches:
- main
pull_request:
branches:
- main
env:
# Customize the CMake build type here (Release, Debug, RelWithDebInfo, etc.)
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_ENABLEAVX: 1
jobs:
hostrunner_tests:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, x64]]
fail-fast: false
steps:
- uses: actions/checkout@v3
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: 'actions/upload-artifact@v3'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
+3 -41
View File
@@ -16,11 +16,11 @@ env:
FEX_ENABLEAVX: 1
jobs:
instcountci_tests:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, x64], [self-hosted, ARM64]]
arch: [[self-hosted, ARM64]]
fail-fast: false
steps:
@@ -56,16 +56,6 @@ jobs:
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Set vixl_sim x86
if: matrix.arch[1] == 'x64'
run: |
echo "VIXL_SIM_ENABLED=True" >> $GITHUB_ENV
- name: Set vixl_sim Arm64
if: matrix.arch[1] == 'ARM64'
run: |
echo "VIXL_SIM_ENABLED=False" >> $GITHUB_ENV
- name: Configure CMake
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
@@ -74,7 +64,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=False -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -96,25 +86,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_InstCountCI.log || true
- name: Update local repo instcount
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: cmake --build . --config $BUILD_TYPE --target instcountci_update_tests
- name: Get instcountCI diff
if: ${{ always() }}
shell: bash
working-directory: ${{github.workspace}}/
run: git diff --output=${{runner.workspace}}/build/InstCountCI.diff
- name: Check if InstCountCI Diff exists
if: ${{ always() }}
shell: bash
working-directory: ${{github.workspace}}/
# Check if the file is empty
run: sh -c "! test -s ${{runner.workspace}}/build/InstCountCI.diff"
- name: Truncate test results
if: ${{ always() }}
shell: bash
@@ -136,12 +107,3 @@ jobs:
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
- name: Upload results InstCountCI
if: ${{ always() }}
uses: 'actions/upload-artifact@v3'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}-instcountci
path: ${{runner.workspace}}/build/InstCountCI.diff
retention-days: 3
+3 -3
View File
@@ -13,11 +13,11 @@ env:
FEX_ENABLEAVX: 1
jobs:
mingw_build:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARM64, mingw]]
arch: [[self-hosted, x64, mingw], [self-hosted, ARM64, mingw]]
fail-fast: false
steps:
@@ -74,7 +74,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=False -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
+1 -1
View File
@@ -16,7 +16,7 @@ env:
FEX_ENABLEAVX: 1
jobs:
vixl_simulator:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
-5
View File
@@ -1,5 +0,0 @@
{
"ThunksDB": {
"fex_thunk_test": 1
}
}
+21 -8
View File
@@ -25,6 +25,7 @@ option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enables jemalloc glibc allocator" TRUE)
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
@@ -96,6 +97,11 @@ if (ENABLE_GDB_SYMBOLS)
endif()
if (ENABLE_INTERPRETER)
message(STATUS "Interpreter enabled")
add_definitions(-DINTERPRETER_ENABLED=1)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
@@ -112,6 +118,14 @@ else()
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
message(FATAL_ERROR
" Be warned: FEX isn't optimized for x86_64 hosts!\n"
" Support for x86_64 hosts is only for debugging and convenience!\n"
" Don't expect amazing performance or optimal code generation!\n"
" Pass -DENABLE_X86_HOST_DEBUG=True to bypass this message!")
endif()
set(_M_X86_64 1)
add_definitions(-D_M_X86_64=1)
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
@@ -293,11 +307,10 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
endif()
endif()
set(FEX_TUNE_COMPILE_FLAGS)
if (NOT TUNE_ARCH STREQUAL "generic")
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
if(COMPILER_SUPPORTS_ARCH_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH}")
add_compile_options("-march=${TUNE_ARCH}")
else()
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
endif()
@@ -310,7 +323,7 @@ if (TUNE_CPU STREQUAL "native")
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=native")
add_compile_options("-mcpu=native")
endif()
else()
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
@@ -324,19 +337,19 @@ if (TUNE_CPU STREQUAL "native")
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${AARCH64_CPU}")
add_compile_options("-mcpu=${AARCH64_CPU}")
endif()
endif()
else()
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
if(COMPILER_SUPPORTS_MARCH_NATIVE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=native")
add_compile_options("-march=native")
endif()
endif()
else()
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${TUNE_CPU}")
add_compile_options("-mcpu=${TUNE_CPU}")
else()
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
endif()
@@ -454,10 +467,10 @@ if (BUILD_THUNKS)
CMAKE_ARGS
"-DBITNESS=64"
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DBUILD_FEX_LINUX_TESTS=${BUILD_FEX_LINUX_TESTS}"
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_64_TOOLCHAIN_FILE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
INSTALL_COMMAND ""
@@ -472,10 +485,10 @@ if (BUILD_THUNKS)
CMAKE_ARGS
"-DBITNESS=32"
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DBUILD_FEX_LINUX_TESTS=${BUILD_FEX_LINUX_TESTS}"
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_32_TOOLCHAIN_FILE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
INSTALL_COMMAND ""
+5
View File
@@ -0,0 +1,5 @@
{
"Config": {
"Env": "STEAM_GAME_LAUNCH_SHELL=@CMAKE_INSTALL_PREFIX@/bin/FEXBash"
}
}
-6
View File
@@ -144,12 +144,6 @@
"@PREFIX_LIB@/libasound.so.2.0.0"
]
},
"fex_thunk_test": {
"Library": "libfex_thunk_test-guest.so",
"Overlay": [
"@PREFIX_LIB@/libfex_thunk_test.so"
]
},
"Xrender": {
"Library": "libXrender-guest.so",
"Overlay": [
+13
View File
@@ -0,0 +1,13 @@
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
Version 2, December 2004
Copyright (C) 2018 Ryan Houdek <Sonicadvance1@gmail.com>
Everyone is permitted to copy and distribute verbatim or modified
copies of this license document, and changing it is allowed as long
as the name is changed.
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION
0. You just DO WHAT THE FUCK YOU WANT TO.
+1 -1
+9 -21
View File
@@ -13,6 +13,15 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
set(_M_ARM_64 1)
endif()
if (ENABLE_VIXL_SIMULATOR)
# If the vixl simulator is enabled then we are using the ARM64 JIT
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" FALSE)
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" TRUE)
else()
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" ${_M_X86_64})
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" ${_M_ARM_64})
endif()
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
@@ -24,27 +33,6 @@ set(CMAKE_INCLUDE_CURRENT_DIR ON)
include(CheckCXXCompilerFlag)
include(CheckIncludeFileCXX)
include(CheckCXXSourceCompiles)
set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
check_cxx_source_compiles(
"
__attribute__((preserve_all))
void Testy() {
}
int main() {
return 0;
}"
HAS_CLANG_PRESERVE_ALL)
unset(CMAKE_REQUIRED_FLAGS)
if (HAS_CLANG_PRESERVE_ALL)
if (MINGW_BUILD)
message(STATUS "Ignoring broken clang::preserve_all support")
set(HAS_CLANG_PRESERVE_ALL FALSE)
else()
message(STATUS "Has clang::preserve_all")
endif()
endif ()
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
# Useful to have for freestanding libFEXCore
+10 -73
View File
@@ -46,15 +46,11 @@ class OpDefinition:
NumElements: str
OpClass: str
HasSideEffects: bool
ImplicitFlagClobber: bool
RAOverride: int
SwitchGen: bool
ArgPrinter: bool
SSAArgNum: int
NonSSAArgNum: int
DynamicDispatch: bool
JITDispatch: bool
JITDispatchOverride: str
Arguments: list
EmitValidation: list
Desc: list
@@ -68,15 +64,11 @@ class OpDefinition:
self.OpClass = None
self.OpSize = 0
self.HasSideEffects = False
self.ImplicitFlagClobber = False
self.RAOverride = -1
self.SwitchGen = True
self.ArgPrinter = True
self.SSAArgNum = 0
self.NonSSAArgNum = 0
self.DynamicDispatch = False
self.JITDispatch = True
self.JITDispatchOverride = None
self.Arguments = []
self.EmitValidation = []
self.Desc = []
@@ -152,7 +144,7 @@ def parse_ops(ops):
Argument = Argument.strip()
OpArg = OpArgument()
Split = Argument.split(":", 1)
Split = Argument.split(":")
if len(Split) != 2:
ExitError("Error parsing argument. Missing Type and name colon split")
@@ -221,9 +213,6 @@ def parse_ops(ops):
if "HasSideEffects" in op_val:
OpDef.HasSideEffects = bool(op_val["HasSideEffects"])
if "ImplicitFlagClobber" in op_val:
OpDef.ImplicitFlagClobber = bool(op_val["ImplicitFlagClobber"])
if "ArgPrinter" in op_val:
OpDef.ArgPrinter = bool(op_val["ArgPrinter"])
@@ -239,15 +228,6 @@ def parse_ops(ops):
if "Desc" in op_val:
OpDef.Desc = op_val["Desc"]
if "DynamicDispatch" in op_val:
OpDef.DynamicDispatch = bool(op_val["DynamicDispatch"])
if "JITDispatch" in op_val:
OpDef.JITDispatch = bool(op_val["JITDispatch"])
if "JITDispatchOverride" in op_val:
OpDef.JITDispatchOverride = op_val["JITDispatchOverride"]
# Do some fixups of the data here
if len(OpDef.EmitValidation) != 0:
for i in range(len(OpDef.EmitValidation)):
@@ -377,7 +357,6 @@ def print_ir_sizes():
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool ImplicitFlagClobber(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
output_file.write("#undef IROP_SIZES\n")
@@ -471,17 +450,15 @@ def print_ir_getraargs():
def print_ir_hassideeffects():
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
for array, prop in [("SideEffects", "HasSideEffects"),
("ImplicitFlagClobbers", "ImplicitFlagClobber")]:
output_file.write(f"constexpr std::array<uint8_t, OP_LAST + 1> {array} = {{\n")
for op in IROps:
output_file.write("\t{},\n".format(("true" if getattr(op, prop) else "false")))
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> SideEffects = {\n")
for op in IROps:
output_file.write("\t{},\n".format(("true" if op.HasSideEffects else "false")))
output_file.write("};\n\n")
output_file.write("};\n\n")
output_file.write(f"bool {prop}(IROps Op) {{\n")
output_file.write(f" return {array}[Op];\n")
output_file.write("}\n")
output_file.write("bool HasSideEffects(IROps Op) {\n")
output_file.write(" return SideEffects[Op];\n")
output_file.write("}\n")
output_file.write("#undef IROP_HASSIDEEFFECTS_IMPL\n")
output_file.write("#endif\n\n")
@@ -650,10 +627,6 @@ def print_ir_allocator_helpers():
output_file.write(") {\n")
# Save NZCV if needed before clobbering NZCV
if op.ImplicitFlagClobber:
output_file.write("\t\tSaveNZCV(IROps::OP_{});".format(op.Name.upper()))
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
if op.SSAArgNum != 0:
@@ -702,8 +675,7 @@ def print_ir_allocator_helpers():
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
for Validation in op.EmitValidation:
Sanitized = Validation.replace("\"", "\\\"")
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"\");\n".format(Validation))
output_file.write("\t\t#endif\n")
output_file.write("\t\treturn Op;\n")
@@ -758,38 +730,10 @@ def print_ir_parser_switch_helper():
output_file.write("#undef IROP_PARSER_SWITCH_HELPERS\n")
output_file.write("#endif\n")
def print_ir_dispatcher_defs():
output_dispatch_file.write("#ifdef IROP_DISPATCH_DEFS\n")
for op in IROps:
if op.Name != "Last" and op.SwitchGen and op.JITDispatch and op.JITDispatchOverride == None:
output_dispatch_file.write("DEF_OP({});\n".format(op.Name))
output_dispatch_file.write("#undef IROP_DISPATCH_DEFS\n")
output_dispatch_file.write("#endif\n")
def print_ir_dispatcher_dispatch():
output_dispatch_file.write("#ifdef IROP_DISPATCH_DISPATCH\n")
for op in IROps:
if op.Name != "Last" and op.JITDispatch:
DispatchName = op.Name
if op.JITDispatchOverride != None:
DispatchName = op.JITDispatchOverride
if (op.DynamicDispatch):
output_dispatch_file.write("REGISTER_OP_RT({}, {});\n".format(op.Name.upper(), DispatchName))
else:
output_dispatch_file.write("REGISTER_OP({}, {});\n".format(op.Name.upper(), DispatchName))
output_dispatch_file.write("#undef IROP_DISPATCH_DISPATCH\n")
output_dispatch_file.write("#endif\n")
if (len(sys.argv) < 4):
if (len(sys.argv) < 3):
ExitError()
output_filename = sys.argv[2]
output_dispatcher_filename = sys.argv[3]
json_file = open(sys.argv[1], "r")
json_text = json_file.read()
json_file.close()
@@ -819,10 +763,3 @@ print_ir_allocator_helpers()
print_ir_parser_switch_helper()
output_file.close()
output_dispatch_file = open(output_dispatcher_filename, "w")
print_ir_dispatcher_defs()
print_ir_dispatcher_dispatch()
output_dispatch_file.close()
+59 -25
View File
@@ -90,6 +90,7 @@ set (SRCS
Interface/Core/CPUBackend.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/GdbServer.cpp
Interface/Core/HostFeatures.cpp
Interface/Core/ObjectCache/JobHandling.cpp
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
@@ -100,23 +101,15 @@ set (SRCS
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher/X87F64.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/SignalDelegator.cpp
Interface/Core/X86Tables.cpp
Interface/Core/X86DebugInfo.cpp
Interface/Core/X86HelperGen.cpp
Interface/Core/ArchHelpers/Arm64Emitter.cpp
Interface/Core/Dispatcher/Dispatcher.cpp
Interface/Core/Dispatcher/X86Dispatcher.cpp
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
Interface/Core/JIT/Arm64/JIT.cpp
Interface/Core/JIT/Arm64/ALUOps.cpp
Interface/Core/JIT/Arm64/AtomicOps.cpp
Interface/Core/JIT/Arm64/BranchOps.cpp
Interface/Core/JIT/Arm64/ConversionOps.cpp
Interface/Core/JIT/Arm64/EncryptionOps.cpp
Interface/Core/JIT/Arm64/FlagOps.cpp
Interface/Core/JIT/Arm64/MemoryOps.cpp
Interface/Core/JIT/Arm64/MiscOps.cpp
Interface/Core/JIT/Arm64/MoveOps.cpp
Interface/Core/JIT/Arm64/VectorOps.cpp
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
Interface/Core/X86Tables/BaseTables.cpp
Interface/Core/X86Tables/DDDTables.cpp
Interface/Core/X86Tables/EVEXTables.cpp
@@ -148,7 +141,7 @@ set (SRCS
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
Interface/IR/Passes/DeadStoreElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/InlineCallOptimization.cpp
Interface/IR/Passes/SyscallOptimization.cpp
Utils/NetStream.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
@@ -166,7 +159,24 @@ if (ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
Utils/AllocatorOverride.cpp)
endif()
set(DEFINES -DTHREAD_LOCAL=_Thread_local -DJIT_ARM64)
if (ENABLE_INTERPRETER)
list(APPEND SRCS
Interface/Core/Interpreter/InterpreterCore.cpp
Interface/Core/Interpreter/InterpreterOps.cpp
Interface/Core/Interpreter/ALUOps.cpp
Interface/Core/Interpreter/AtomicOps.cpp
Interface/Core/Interpreter/BranchOps.cpp
Interface/Core/Interpreter/ConversionOps.cpp
Interface/Core/Interpreter/EncryptionOps.cpp
Interface/Core/Interpreter/F80Ops.cpp
Interface/Core/Interpreter/FlagOps.cpp
Interface/Core/Interpreter/MemoryOps.cpp
Interface/Core/Interpreter/MiscOps.cpp
Interface/Core/Interpreter/MoveOps.cpp
Interface/Core/Interpreter/VectorOps.cpp)
endif()
set(DEFINES -DTHREAD_LOCAL=_Thread_local)
if (_M_X86_64)
list(APPEND DEFINES -D_M_X86_64=1)
@@ -185,14 +195,41 @@ if (ENABLE_VIXL_DISASSEMBLER)
list(APPEND DEFINES -DVIXL_DISASSEMBLER=1)
endif()
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=__attribute__((preserve_all));-DFEXCORE_HAS_PRESERVE_ALL_ATTR=1")
else()
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=;-DFEXCORE_HAS_PRESERVE_ALL_ATTR=0")
if (ENABLE_JIT_X86_64)
list(APPEND SRCS
Interface/Core/JIT/x86_64/JIT.cpp
Interface/Core/JIT/x86_64/ALUOps.cpp
Interface/Core/JIT/x86_64/AtomicOps.cpp
Interface/Core/JIT/x86_64/BranchOps.cpp
Interface/Core/JIT/x86_64/ConversionOps.cpp
Interface/Core/JIT/x86_64/EncryptionOps.cpp
Interface/Core/JIT/x86_64/FlagOps.cpp
Interface/Core/JIT/x86_64/MemoryOps.cpp
Interface/Core/JIT/x86_64/MiscOps.cpp
Interface/Core/JIT/x86_64/MoveOps.cpp
Interface/Core/JIT/x86_64/VectorOps.cpp
Interface/Core/JIT/x86_64/x64Relocations.cpp
)
list(APPEND DEFINES -DJIT_X86_64)
endif()
# Some defines for the softfloat library
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
if (ENABLE_JIT_ARM64)
list(APPEND DEFINES -DJIT_ARM64)
list(APPEND SRCS
Interface/Core/JIT/Arm64/JIT.cpp
Interface/Core/JIT/Arm64/ALUOps.cpp
Interface/Core/JIT/Arm64/AtomicOps.cpp
Interface/Core/JIT/Arm64/BranchOps.cpp
Interface/Core/JIT/Arm64/ConversionOps.cpp
Interface/Core/JIT/Arm64/EncryptionOps.cpp
Interface/Core/JIT/Arm64/FlagOps.cpp
Interface/Core/JIT/Arm64/MemoryOps.cpp
Interface/Core/JIT/Arm64/MiscOps.cpp
Interface/Core/JIT/Arm64/MoveOps.cpp
Interface/Core/JIT/Arm64/VectorOps.cpp
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
)
endif()
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
@@ -218,16 +255,15 @@ configure_file(
# Generate IR include file
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
set(OUTPUT_DISPATCHER_NAME "${OUTPUT_IR_FOLDER}/IRDefines_Dispatch.inc")
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
add_custom_command(
OUTPUT "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
OUTPUT "${OUTPUT_NAME}"
DEPENDS "${INPUT_NAME}"
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
)
set_source_files_properties(${OUTPUT_NAME} PROPERTIES
@@ -359,7 +395,6 @@ function(AddObject Name Type)
add_library(${Name} ${Type} ${SRCS})
target_link_libraries(${Name} FEXCore_Base)
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
AddDefaultOptionsToTarget(${Name})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
@@ -368,7 +403,6 @@ endfunction()
function(AddLibrary Name Type)
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
target_link_libraries(${Name} FEXCore_Base)
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
if (MINGW_BUILD)
# Mingw build isn't building a linux shared library, so it can't have a SONAME.
-1
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/Allocator.h>
+36 -86
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/fextl/fmt.h>
#include "Common/JitSymbols.h"
@@ -27,6 +26,42 @@ namespace FEXCore {
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
}
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
@@ -51,89 +86,4 @@ namespace FEXCore {
}
}
// Buffered JIT symbols.
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, GuestAddr, CodeSize);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, CodeSize, Name, Offset);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}\n", HostAddr, CodeSize, Name);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
RegisterNamedRegion(Buffer, HostAddr, CodeSize, Name);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite) {
auto Now = std::chrono::steady_clock::now();
if (!ForceWrite) {
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) &&
Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
// Still buffering, no need to write.
return;
}
}
Buffer->LastWrite = Now;
auto Result = write(fd, Buffer->Buffer, Buffer->Offset);
if (Result == -1 && errno == EBADF) {
fd = -1;
}
Buffer->Offset = 0;
}
} // namespace FEXCore
+3 -15
View File
@@ -1,10 +1,5 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/fextl/memory.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <chrono>
#include <cstdint>
#include <cstdio>
#include <memory>
@@ -17,20 +12,13 @@ public:
~JITSymbols();
void InitFile();
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
// Allocate JIT buffer.
static fextl::unique_ptr<Core::JITSymbolBuffer> AllocateBuffer() {
return fextl::make_unique<Core::JITSymbolBuffer>();
}
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name);
private:
int fd{-1};
void WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite = false);
};
}
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_add( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_div( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
bool extF80_eq( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
bool extF80_lt( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_mul( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_rem( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
extF80_roundToInt( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_sqrt( extFloat80_t a )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_sub( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float128_t extF80_to_f128( extFloat80_t a )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float32_t extF80_to_f32( extFloat80_t a )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float64_t extF80_to_f64( extFloat80_t a )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
int_fast32_t
extF80_to_i32( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
int_fast64_t
extF80_to_i64( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
uint_fast64_t
extF80_to_ui64( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f128_to_extF80( float128_t a )
{
union ui128_f128 uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f32_to_extF80( float32_t a )
{
union ui32_f32 uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f64_to_extF80( float64_t a )
{
union ui64_f64 uA;
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t i32_to_extF80( int32_t a )
{
uint_fast16_t uiZ64;
@@ -68,11 +68,9 @@ uint_fast64_t
uint_fast64_t softfloat_roundMToUI64( bool, uint32_t *, uint_fast8_t, bool );
#endif
FEXCORE_PRESERVE_ALL_ATTR
int_fast32_t softfloat_roundToI32( bool, uint_fast64_t, uint_fast8_t, bool );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
int_fast64_t
softfloat_roundToI64(
bool, uint_fast64_t, uint_fast64_t, uint_fast8_t, bool );
@@ -111,10 +109,8 @@ float16_t
#define isNaNF32UI( a ) (((~(a) & 0x7F800000) == 0) && ((a) & 0x007FFFFF))
struct exp16_sig32 { int_fast16_t exp; uint_fast32_t sig; };
FEXCORE_PRESERVE_ALL_ATTR
struct exp16_sig32 softfloat_normSubnormalF32Sig( uint_fast32_t );
FEXCORE_PRESERVE_ALL_ATTR
float32_t softfloat_roundPackToF32( bool, int_fast16_t, uint_fast32_t );
float32_t softfloat_normRoundPackToF32( bool, int_fast16_t, uint_fast32_t );
@@ -134,10 +130,8 @@ float32_t
#define isNaNF64UI( a ) (((~(a) & UINT64_C( 0x7FF0000000000000 )) == 0) && ((a) & UINT64_C( 0x000FFFFFFFFFFFFF )))
struct exp16_sig64 { int_fast16_t exp; uint_fast64_t sig; };
FEXCORE_PRESERVE_ALL_ATTR
struct exp16_sig64 softfloat_normSubnormalF64Sig( uint_fast64_t );
FEXCORE_PRESERVE_ALL_ATTR
float64_t softfloat_roundPackToF64( bool, int_fast16_t, uint_fast64_t );
float64_t softfloat_normRoundPackToF64( bool, int_fast16_t, uint_fast64_t );
@@ -161,14 +155,11 @@ float64_t
*----------------------------------------------------------------------------*/
struct exp32_sig64 { int_fast32_t exp; uint64_t sig; };
FEXCORE_PRESERVE_ALL_ATTR
struct exp32_sig64 softfloat_normSubnormalExtF80Sig( uint_fast64_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
softfloat_roundPackToExtF80(
bool, int_fast32_t, uint_fast64_t, uint_fast64_t, uint_fast8_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
softfloat_normRoundPackToExtF80(
bool, int_fast32_t, uint_fast64_t, uint_fast64_t, uint_fast8_t );
@@ -190,7 +181,6 @@ extFloat80_t
#define isNaNF128UI( a64, a0 ) (((~(a64) & UINT64_C( 0x7FFF000000000000 )) == 0) && (a0 || ((a64) & UINT64_C( 0x0000FFFFFFFFFFFF ))))
struct exp32_sig128 { int_fast32_t exp; struct uint128 sig; };
FEXCORE_PRESERVE_ALL_ATTR
struct exp32_sig128
softfloat_normSubnormalF128Sig( uint_fast64_t, uint_fast64_t );
@@ -53,7 +53,6 @@ INLINE
uint64_t softfloat_shortShiftRightJam64( uint64_t a, uint_fast8_t dist )
{ return a>>dist | ((a & (((uint_fast64_t) 1<<dist) - 1)) != 0); }
#else
FEXCORE_PRESERVE_ALL_ATTR
uint64_t softfloat_shortShiftRightJam64( uint64_t a, uint_fast8_t dist );
#endif
#endif
@@ -75,7 +74,6 @@ INLINE uint32_t softfloat_shiftRightJam32( uint32_t a, uint_fast16_t dist )
(dist < 31) ? a>>dist | ((uint32_t) (a<<(-dist & 31)) != 0) : (a != 0);
}
#else
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_shiftRightJam32( uint32_t a, uint_fast16_t dist );
#endif
#endif
@@ -97,7 +95,6 @@ INLINE uint64_t softfloat_shiftRightJam64( uint64_t a, uint_fast32_t dist )
(dist < 63) ? a>>dist | ((uint64_t) (a<<(-dist & 63)) != 0) : (a != 0);
}
#else
FEXCORE_PRESERVE_ALL_ATTR
uint64_t softfloat_shiftRightJam64( uint64_t a, uint_fast32_t dist );
#endif
#endif
@@ -151,7 +148,6 @@ INLINE uint_fast8_t softfloat_countLeadingZeros32( uint32_t a )
return count;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
uint_fast8_t softfloat_countLeadingZeros32( uint32_t a );
#endif
#endif
@@ -161,7 +157,6 @@ uint_fast8_t softfloat_countLeadingZeros32( uint32_t a );
| Returns the number of leading 0 bits before the most-significant 1 bit of
| 'a'. If 'a' is zero, 64 is returned.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast8_t softfloat_countLeadingZeros64( uint64_t a );
#endif
@@ -183,7 +178,6 @@ extern const uint16_t softfloat_approxRecip_1k1s[16];
#ifdef SOFTFLOAT_FAST_DIV64TO32
#define softfloat_approxRecip32_1( a ) ((uint32_t) (UINT64_C( 0x7FFFFFFFFFFFFFFF ) / (uint32_t) (a)))
#else
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_approxRecip32_1( uint32_t a );
#endif
#endif
@@ -210,7 +204,6 @@ extern const uint16_t softfloat_approxRecipSqrt_1k1s[16];
| returned is also always within the range 0.5 to 1; thus, the most-
| significant bit of the result is always set.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_approxRecipSqrt32_1( unsigned int oddExpA, uint32_t a );
#endif
@@ -247,7 +240,6 @@ INLINE
bool softfloat_le128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{ return (a64 < b64) || ((a64 == b64) && (a0 <= b0)); }
#else
FEXCORE_PRESERVE_ALL_ATTR
bool softfloat_le128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 );
#endif
#endif
@@ -263,7 +255,6 @@ INLINE
bool softfloat_lt128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{ return (a64 < b64) || ((a64 == b64) && (a0 < b0)); }
#else
FEXCORE_PRESERVE_ALL_ATTR
bool softfloat_lt128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 );
#endif
#endif
@@ -284,7 +275,6 @@ struct uint128
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_shortShiftLeft128( uint64_t a64, uint64_t a0, uint_fast8_t dist );
#endif
@@ -306,7 +296,6 @@ struct uint128
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_shortShiftRight128( uint64_t a64, uint64_t a0, uint_fast8_t dist );
#endif
@@ -424,7 +413,6 @@ struct uint64_extra
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint64_extra
softfloat_shiftRightJam64Extra(
uint64_t a, uint64_t extra, uint_fast32_t dist );
@@ -504,7 +492,6 @@ struct uint128
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_add128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 );
#endif
@@ -541,7 +528,6 @@ struct uint128
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_sub128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 );
#endif
@@ -576,7 +562,6 @@ INLINE struct uint128 softfloat_mul64ByShifted32To128( uint64_t a, uint32_t b )
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_mul64ByShifted32To128( uint64_t a, uint32_t b );
#endif
#endif
@@ -585,7 +570,6 @@ struct uint128 softfloat_mul64ByShifted32To128( uint64_t a, uint32_t b );
/*----------------------------------------------------------------------------
| Returns the 128-bit product of 'a' and 'b'.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_mul64To128( uint64_t a, uint64_t b );
#endif
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_add128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_add128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
extern const uint16_t softfloat_approxRecip_1k0s[16];
extern const uint16_t softfloat_approxRecip_1k1s[16];
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_approxRecip32_1( uint32_t a )
{
int index;
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
extern const uint16_t softfloat_approxRecipSqrt_1k0s[];
extern const uint16_t softfloat_approxRecipSqrt_1k1s[];
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_approxRecipSqrt32_1( unsigned int oddExpA, uint32_t a )
{
int index;
@@ -44,7 +44,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| floating-point NaN, and returns the bit pattern of this value as an unsigned
| integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_commonNaNToExtF80UI( const struct commonNaN *aPtr )
{
struct uint128 uiZ;
@@ -43,7 +43,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| Converts the common NaN pointed to by `aPtr' into a 128-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_commonNaNToF128UI( const struct commonNaN *aPtr )
{
struct uint128 uiZ;
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| Converts the common NaN pointed to by `aPtr' into a 32-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast32_t softfloat_commonNaNToF32UI( const struct commonNaN *aPtr )
{
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| Converts the common NaN pointed to by `aPtr' into a 64-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast64_t softfloat_commonNaNToF64UI( const struct commonNaN *aPtr )
{
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#define softfloat_countLeadingZeros32 softfloat_countLeadingZeros32
#include "primitives.h"
FEXCORE_PRESERVE_ALL_ATTR
uint_fast8_t softfloat_countLeadingZeros32( uint32_t a )
{
uint_fast8_t count;
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#define softfloat_countLeadingZeros64 softfloat_countLeadingZeros64
#include "primitives.h"
FEXCORE_PRESERVE_ALL_ATTR
uint_fast8_t softfloat_countLeadingZeros64( uint64_t a )
{
uint_fast8_t count;
@@ -46,7 +46,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| location pointed to by `zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void
softfloat_extF80UIToCommonNaN(
uint_fast16_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr )
@@ -47,7 +47,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| pointed to by `zPtr'. If the NaN is a signaling NaN, the invalid exception
| is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void
softfloat_f128UIToCommonNaN(
uint_fast64_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr )
@@ -45,7 +45,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| location pointed to by `zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_f32UIToCommonNaN( uint_fast32_t uiA, struct commonNaN *zPtr )
{
@@ -45,7 +45,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| location pointed to by `zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_f64UIToCommonNaN( uint_fast64_t uiA, struct commonNaN *zPtr )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_le128
FEXCORE_PRESERVE_ALL_ATTR
bool softfloat_le128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_lt128
FEXCORE_PRESERVE_ALL_ATTR
bool softfloat_lt128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_mul64ByShifted32To128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_mul64ByShifted32To128( uint64_t a, uint32_t b )
{
uint_fast64_t mid;
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_mul64To128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_mul64To128( uint64_t a, uint64_t b )
{
uint32_t a32, a0, b32, b0;
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
softfloat_normRoundPackToExtF80(
bool sign,
@@ -38,7 +38,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
struct exp32_sig64 softfloat_normSubnormalExtF80Sig( uint_fast64_t sig )
{
int_fast8_t shiftDist;
@@ -38,7 +38,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
struct exp32_sig128
softfloat_normSubnormalF128Sig( uint_fast64_t sig64, uint_fast64_t sig0 )
{
@@ -38,7 +38,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
struct exp16_sig32 softfloat_normSubnormalF32Sig( uint_fast32_t sig )
{
int_fast8_t shiftDist;
@@ -38,7 +38,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
struct exp16_sig64 softfloat_normSubnormalF64Sig( uint_fast64_t sig )
{
int_fast8_t shiftDist;
@@ -50,7 +50,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| result. If either original floating-point value is a signaling NaN, the
| invalid exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_propagateNaNExtF80UI(
uint_fast16_t uiA64,
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
softfloat_roundPackToExtF80(
bool sign,
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float32_t
softfloat_roundPackToF32( bool sign, int_fast16_t exp, uint_fast32_t sig )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float64_t
softfloat_roundPackToF64( bool sign, int_fast16_t exp, uint_fast64_t sig )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
int_fast32_t
softfloat_roundToI32(
bool sign, uint_fast64_t sig, uint_fast8_t roundingMode, bool exact )
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
int_fast64_t
softfloat_roundToI64(
bool sign,
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shiftRightJam32
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_shiftRightJam32( uint32_t a, uint_fast16_t dist )
{
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shiftRightJam64
FEXCORE_PRESERVE_ALL_ATTR
uint64_t softfloat_shiftRightJam64( uint64_t a, uint_fast32_t dist )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shiftRightJam64Extra
FEXCORE_PRESERVE_ALL_ATTR
struct uint64_extra
softfloat_shiftRightJam64Extra(
uint64_t a, uint64_t extra, uint_fast32_t dist )
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shortShiftLeft128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_shortShiftLeft128( uint64_t a64, uint64_t a0, uint_fast8_t dist )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shortShiftRight128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_shortShiftRight128( uint64_t a64, uint64_t a0, uint_fast8_t dist )
{
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shortShiftRightJam64
FEXCORE_PRESERVE_ALL_ATTR
uint64_t softfloat_shortShiftRightJam64( uint64_t a, uint_fast8_t dist )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_sub128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_sub128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{
@@ -92,7 +92,6 @@ enum {
/*----------------------------------------------------------------------------
| Routine to raise any or all of the software floating-point exception flags.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_raiseFlags( uint_fast8_t );
/*----------------------------------------------------------------------------
@@ -111,7 +110,6 @@ float16_t ui64_to_f16( uint64_t );
float32_t ui64_to_f32( uint64_t );
float64_t ui64_to_f64( uint64_t );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t ui64_to_extF80( uint64_t );
float128_t ui64_to_f128( uint64_t );
#endif
@@ -121,7 +119,6 @@ float16_t i32_to_f16( int32_t );
float32_t i32_to_f32( int32_t );
float64_t i32_to_f64( int32_t );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t i32_to_extF80( int32_t );
float128_t i32_to_f128( int32_t );
#endif
@@ -186,7 +183,6 @@ int_fast64_t f32_to_i64_r_minMag( float32_t, bool );
float16_t f32_to_f16( float32_t );
float64_t f32_to_f64( float32_t );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f32_to_extF80( float32_t );
float128_t f32_to_f128( float32_t );
#endif
@@ -222,7 +218,6 @@ int_fast64_t f64_to_i64_r_minMag( float64_t, bool );
float16_t f64_to_f16( float64_t );
float32_t f64_to_f32( float64_t );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f64_to_extF80( float64_t );
float128_t f64_to_f128( float64_t );
#endif
@@ -255,41 +250,26 @@ extern THREAD_LOCAL uint_fast8_t extF80_roundingPrecision;
*----------------------------------------------------------------------------*/
#ifdef SOFTFLOAT_FAST_INT64
uint_fast32_t extF80_to_ui32( extFloat80_t, uint_fast8_t, bool );
FEXCORE_PRESERVE_ALL_ATTR
uint_fast64_t extF80_to_ui64( extFloat80_t, uint_fast8_t, bool );
FEXCORE_PRESERVE_ALL_ATTR
int_fast32_t extF80_to_i32( extFloat80_t, uint_fast8_t, bool );
FEXCORE_PRESERVE_ALL_ATTR
int_fast64_t extF80_to_i64( extFloat80_t, uint_fast8_t, bool );
uint_fast32_t extF80_to_ui32_r_minMag( extFloat80_t, bool );
uint_fast64_t extF80_to_ui64_r_minMag( extFloat80_t, bool );
int_fast32_t extF80_to_i32_r_minMag( extFloat80_t, bool );
int_fast64_t extF80_to_i64_r_minMag( extFloat80_t, bool );
float16_t extF80_to_f16( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
float32_t extF80_to_f32( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
float64_t extF80_to_f64( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
float128_t extF80_to_f128( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_roundToInt( extFloat80_t, uint_fast8_t, bool );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_add( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_sub( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_mul( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_div( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_rem( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_sqrt( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
bool extF80_eq( extFloat80_t, extFloat80_t );
bool extF80_le( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
bool extF80_lt( extFloat80_t, extFloat80_t );
bool extF80_eq_signaling( extFloat80_t, extFloat80_t );
bool extF80_le_quiet( extFloat80_t, extFloat80_t );
@@ -340,7 +320,6 @@ int_fast64_t f128_to_i64_r_minMag( float128_t, bool );
float16_t f128_to_f16( float128_t );
float32_t f128_to_f32( float128_t );
float64_t f128_to_f64( float128_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f128_to_extF80( float128_t );
float128_t f128_roundToInt( float128_t, uint_fast8_t, bool );
float128_t f128_add( float128_t, float128_t );
@@ -43,7 +43,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| to substitute a result value. If traps are not implemented, this routine
| should be simply `softfloat_exceptionFlags |= flags;'.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_raiseFlags( uint_fast8_t flags )
{
@@ -135,14 +135,12 @@ uint_fast16_t
| location pointed to by 'zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_f32UIToCommonNaN( uint_fast32_t uiA, struct commonNaN *zPtr );
/*----------------------------------------------------------------------------
| Converts the common NaN pointed to by 'aPtr' into a 32-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast32_t softfloat_commonNaNToF32UI( const struct commonNaN *aPtr );
/*----------------------------------------------------------------------------
@@ -172,14 +170,12 @@ uint_fast32_t
| location pointed to by 'zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_f64UIToCommonNaN( uint_fast64_t uiA, struct commonNaN *zPtr );
/*----------------------------------------------------------------------------
| Converts the common NaN pointed to by 'aPtr' into a 64-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast64_t softfloat_commonNaNToF64UI( const struct commonNaN *aPtr );
/*----------------------------------------------------------------------------
@@ -219,7 +215,6 @@ uint_fast64_t
| location pointed to by 'zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void
softfloat_extF80UIToCommonNaN(
uint_fast16_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr );
@@ -229,7 +224,6 @@ void
| floating-point NaN, and returns the bit pattern of this value as an unsigned
| integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_commonNaNToExtF80UI( const struct commonNaN *aPtr );
/*----------------------------------------------------------------------------
@@ -241,7 +235,6 @@ struct uint128 softfloat_commonNaNToExtF80UI( const struct commonNaN *aPtr );
| result. If either original floating-point value is a signaling NaN, the
| invalid exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_propagateNaNExtF80UI(
uint_fast16_t uiA64,
@@ -271,7 +264,6 @@ struct uint128
| pointed to by 'zPtr'. If the NaN is a signaling NaN, the invalid exception
| is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void
softfloat_f128UIToCommonNaN(
uint_fast64_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr );
@@ -280,7 +272,6 @@ void
| Converts the common NaN pointed to by 'aPtr' into a 128-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_commonNaNToF128UI( const struct commonNaN * );
/*----------------------------------------------------------------------------
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t ui64_to_extF80( uint64_t a )
{
uint_fast16_t uiZ64;
-20
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/BitUtils.h>
@@ -63,7 +62,6 @@ struct FEX_PACKED X80SoftFloat {
}
// Ops
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
@@ -85,7 +83,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
@@ -107,7 +104,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
@@ -129,7 +125,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
@@ -151,7 +146,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
@@ -174,7 +168,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
@@ -197,17 +190,14 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
return extF80_roundToInt(lhs, RoundMode, false);
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
@@ -231,7 +221,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
@@ -253,14 +242,12 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
*eq = extF80_eq(lhs, rhs);
*lt = extF80_lt(lhs, rhs);
*nan = IsNan(lhs) || IsNan(rhs);
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -289,7 +276,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -313,7 +299,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -339,7 +324,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -365,7 +349,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -389,7 +372,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -412,7 +394,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -435,7 +416,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
-1
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/fextl/string.h>
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/fextl/string.h>
+24 -5
View File
@@ -1,5 +1,5 @@
// SPDX-License-Identifier: MIT
#include "Common/StringConv.h"
#include "Common/StringUtils.h"
#include "FEXCore/Utils/EnumUtils.h"
#include <FEXCore/Config/Config.h>
@@ -7,7 +7,6 @@
#include <FEXCore/Utils/CPUInfo.h>
#include <FEXCore/Utils/FileLoading.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/StringUtils.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/list.h>
#include <FEXCore/fextl/map.h>
@@ -321,15 +320,30 @@ namespace DefaultValues {
Meta->Load();
// Do configuration option fix ups after everything is reloaded
{
// Always fix up the number of threads and create the configuration
// Otherwise the application could receive zero as the number of threads
FEX_CONFIG_OPT(Cores, THREADS);
if (Cores == 0) {
// When the number of emulated CPU cores is zero then auto detect
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, fextl::fmt::format("{}", FEXCore::CPUInfo::CalculateNumberOfCPUs()));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
// Sanitize Core option
FEX_CONFIG_OPT(Core, CORE);
#if (_M_X86_64)
constexpr uint32_t MaxCoreNumber = 1;
constexpr uint32_t MaxCoreNumber = 2;
#else
constexpr uint32_t MaxCoreNumber = 0;
constexpr uint32_t MaxCoreNumber = 1;
#endif
if (Core > MaxCoreNumber) {
#ifdef INTERPRETER_ENABLED
constexpr uint32_t MinCoreNumber = 0;
#else
constexpr uint32_t MinCoreNumber = 1;
#endif
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
// Sanitize the core option by setting the core to the JIT if invalid
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
}
@@ -338,6 +352,11 @@ namespace DefaultValues {
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(Core, CORE);
if (CacheObjectCodeCompilation() && Core() == FEXCore::Config::CONFIG_INTERPRETER) {
// If running the interpreter then disable cache code compilation
FEXCore::Config::Erase(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION);
}
}
fextl::string ContainerPrefix { FindContainerPrefix() };
+13 -28
View File
@@ -6,12 +6,12 @@
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
"TextDefault": "irjit",
"ShortArg": "c",
"Choices": [ "irjit", "host" ],
"Choices": [ "irint", "irjit", "host" ],
"ArgumentHandler": "CoreHandler",
"Desc": [
"Which CPU core to use",
"host only exists on x86_64",
"[irjit, host]"
"[irint, irjit, host]"
]
},
"Multiblock": {
@@ -31,6 +31,15 @@
"Maximum number of instruction to store in a block"
]
},
"Threads": {
"Type": "uint32",
"Default": "0",
"ShortArg": "T",
"Desc": [
"Number of physical hardware threads to tell the process we have.",
"0 will auto detect."
]
},
"CacheObjectCodeCompilation": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
@@ -50,8 +59,6 @@
"DISABLESVE": "disablesve",
"ENABLEAVX": "enableavx",
"DISABLEAVX": "disableavx",
"ENABLEAVX2": "enableavx2",
"DISABLEAVX2": "disableavx2",
"ENABLEAFP": "enableafp",
"DISABLEAFP": "disableafp",
"ENABLELRCPC": "enablelrcpc",
@@ -69,22 +76,13 @@
"ENABLEATOMICS": "enableatomics",
"DISABLEATOMICS": "disableatomics",
"ENABLEFCMA": "enablefcma",
"DISABLEFCMA": "disablefcma",
"ENABLEFLAGM": "enableflagm",
"DISABLEFLAGM": "disableflagm",
"ENABLEFLAGM2": "enableflagm2",
"DISABLEFLAGM2": "disableflagm2",
"ENABLECRYPTO": "enablecrypto",
"DISABLECRYPTO": "disablecrypto",
"ENABLERPRES": "enablerpres",
"DISABLERPRES": "disablerpres"
"DISABLEFCMA": "disablefcma"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
"\toff: Default CPU features queried from CPU features",
"\t{enable,disable}sve: Will force enable or disable sve even if the host doesn't support it",
"\t{enable,disable}avx: Will force enable or disable avx even if the host doesn't support it",
"\t{enable,disable}avx2: Will force enable or disable avx2 even if the host doesn't support it",
"\t{enable,disable}afp: Will force enable or disable afp even if the host doesn't support it",
"\t{enable,disable}lrcpc: Will force enable or disable lrcpc even if the host doesn't support it",
"\t{enable,disable}lrcpc2: Will force enable or disable lrcpc2 even if the host doesn't support it",
@@ -93,11 +91,7 @@
"\t{enable,disable}rng: Will force enable or disable rng even if the host doesn't support it",
"\t{enable,disable}clzero: Will force enable or disable clzero even if the host doesn't support it",
"\t{enable,disable}atomics: Will force enable or disable ARMv8.1 LSE atomics even if the host doesn't support it",
"\t{enable,disable}fcma: Will force enable or disable fcma even if the host doesn't support it",
"\t{enable,disable}flagm: Will force enable or disable flagm even if the host doesn't support it",
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it",
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it"
"\t{enable,disable}fcma: Will force enable or disable fcma even if the host doesn't support it"
]
}
},
@@ -496,15 +490,6 @@
"IS64BIT_MODE": {
"Type": "bool",
"Default": "false"
},
"DISABLE_VIXL_INDIRECT_RUNTIME_CALLS": {
"Type": "bool",
"Default": "true",
"Desc": [
"This option is used for the InstructionCountCI so it can generate the same codegen between Arm64 hosts and vixl simulator hosts.",
"Vixl simulator indirect runtime calls are a special hlt instruction with metadata after it. Effectively making a custom call instruction.",
"With visual simulator calls disabled, the code generation would be the same as on a native Arm64 host, but running the code is broken."
]
}
}
}
+16 -3
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#include "Interface/Context/Context.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
@@ -26,6 +25,12 @@ namespace FEXCore::Context {
return fextl::make_unique<FEXCore::Context::ContextImpl>();
}
bool FEXCore::Context::ContextImpl::InitializeContext() {
// This should be used for generating things that are shared between threads
CPUID.Init(this);
return true;
}
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
CustomExitHandler = std::move(handler);
}
@@ -42,14 +47,22 @@ namespace FEXCore::Context {
CompileBlock(Thread->CurrentFrame, GuestRIP);
}
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
FEXCore::Context::ExitReason FEXCore::Context::ContextImpl::GetExitReason() {
return ParentThread->ExitReason;
}
bool FEXCore::Context::ContextImpl::IsDone() const {
return IsPaused();
}
void FEXCore::Context::ContextImpl::GetCPUState(FEXCore::Core::CPUState *State) const {
memcpy(State, ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
}
void FEXCore::Context::ContextImpl::SetCPUState(const FEXCore::Core::CPUState *State) {
memcpy(ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
}
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
CustomCPUFactory = std::move(Factory);
}
+42 -49
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Common/JitSymbols.h"
@@ -14,8 +13,8 @@
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/DeferredSignalMutex.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
@@ -37,6 +36,7 @@
namespace FEXCore {
class CodeLoader;
class ThunkHandler;
class GdbServer;
namespace CodeSerialize {
class CodeObjectSerializeService;
@@ -45,6 +45,7 @@ namespace CodeSerialize {
namespace CPU {
class Arm64JITCore;
class X86JITCore;
class InterpreterCore;
class Dispatcher;
}
namespace HLE {
@@ -72,6 +73,8 @@ namespace FEXCore::Context {
class ContextImpl final : public FEXCore::Context::Context {
public:
// Context base class implementation.
bool InitializeContext() override;
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
void SetExitHandler(ExitHandler handler) override;
@@ -84,13 +87,17 @@ namespace FEXCore::Context {
ExitReason RunUntilExit() override;
void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) override;
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
int GetProgramStatus() const override;
ExitReason GetExitReason() override;
bool IsDone() const override;
void GetCPUState(FEXCore::Core::CPUState *State) const override;
void SetCPUState(const FEXCore::Core::CPUState *State) override;
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
HostFeatures GetHostFeatures() const override;
@@ -98,37 +105,31 @@ namespace FEXCore::Context {
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) override;
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) override;
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param InitialRIP The starting RIP of this thread
* @param StackPointer The starting RSP of this thread
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
* @param NewThreadState The initial thread state to setup for our state
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* Parent thread Creation:
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
* - CTX->RunUntilExit(Thread);
* OS thread Creation:
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread = CreateThread(NewState, PPID);
* - InitializeThread(Thread);
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
* - Thread = CreateThread(CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread = CreateThread(NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
@@ -203,6 +204,7 @@ namespace FEXCore::Context {
friend class FEXCore::CPU::X86JITCore;
#endif
friend class FEXCore::CPU::InterpreterCore;
friend class FEXCore::IR::Validation::IRValidation;
struct {
@@ -212,9 +214,6 @@ namespace FEXCore::Context {
// this is for internal use
bool ValidateIRarser { false };
// Used if the JIT needs to have its interrupt fault code emitted.
bool NeedsPendingInterruptFaultCheck { false };
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
@@ -241,7 +240,6 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
} Config;
FEXCore::HostFeatures HostFeatures;
@@ -281,18 +279,22 @@ namespace FEXCore::Context {
~ContextImpl();
bool IsPaused() const { return !Running; }
void WaitForThreadsToRun() override;
void WaitForThreadsToRun();
void Stop(bool IgnoreCurrentThread);
void WaitForIdle() override;
void WaitForIdle();
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
void StartGdbServer();
void StopGdbServer();
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
return Fn(Frame, record);
}
@@ -303,7 +305,7 @@ namespace FEXCore::Context {
auto Thread = Frame->Thread;
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
@@ -318,7 +320,7 @@ namespace FEXCore::Context {
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
struct CompileCodeResult {
void* CompiledCode;
@@ -329,8 +331,8 @@ namespace FEXCore::Context {
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
// same as CompileBlock, but aborts on failure
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
@@ -345,6 +347,8 @@ namespace FEXCore::Context {
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
fextl::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
FEXCore::JITSymbols Symbols;
@@ -374,31 +378,10 @@ namespace FEXCore::Context {
UpdateAtomicTSOEmulationConfig();
}
// Returns if Software TSO emulation is required.
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
// This will still return true if on a single thread and TSO is currently disabled.
//
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
// we return consistent results.
//
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
bool SoftwareTSORequired() const {
if (SupportsHardwareTSO) return false;
return Config.TSOEnabled;
}
void EnableExitOnHLT() override { ExitOnHLT = true; }
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
ThreadsState GetThreads() override {
return ThreadsState {
.ParentThread = ParentThread,
.Threads = &Threads,
};
}
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
protected:
@@ -416,6 +399,15 @@ namespace FEXCore::Context {
}
private:
/**
* @brief Does some final thread initialization
*
* @param Thread The internal FEX thread state object
*
* InitCore and CreateThread both call this to finish up thread object initialization
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Initializes the JIT compilers for the thread
*
@@ -433,6 +425,7 @@ namespace FEXCore::Context {
// Entry Cache
std::mutex ExitMutex;
fextl::unique_ptr<GdbServer> DebugServer;
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
@@ -1,9 +1,6 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "FEXCore/Core/X86Enums.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Context/Context.h"
#include "Interface/HLE/Thunks/Thunks.h"
@@ -29,7 +26,7 @@ namespace FEXCore::CPU {
namespace x64 {
// All but x19 and x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
@@ -37,23 +34,23 @@ namespace x64 {
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29,
// PF/AF must be last.
REG_PF, REG_AF,
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
};
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
// All these callee saved
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
FEXCore::ARMEmitter::Reg::r30,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
}};
// All are caller saved
@@ -69,102 +66,11 @@ namespace x64 {
};
// v8..v15 = (lower 64bits) Callee saved
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
// v0 ~ v1 are used as temps.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
// v0 ~ v3 are used as temps.
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
};
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 7> PreserveAll_SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
};
constexpr uint32_t PreserveAll_SRAMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRA) {
switch (Reg.Idx()) {
case 0:
case 1:
case 2:
case 3:
case 4:
case 5:
case 6:
case 7:
case 8:
case 16:
case 17:
Mask |= (1U << Reg.Idx());
break;
default: break;
}
}
return Mask;
}()
};
// Dynamic GPRs
constexpr std::array<FEXCore::ARMEmitter::Register, 1> PreserveAll_Dynamic = {
// Only LR needs to get saved.
FEXCore::ARMEmitter::Reg::r30
};
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
// None.
};
constexpr uint32_t PreserveAll_SRAFPRMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPR) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs
// - v0-v7
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
// v0 ~ v1 are temps
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
};
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
// This is /all/ of the SRA registers
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> PreserveAll_SRAFPRSVE = SRAFPR;
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPRSVE) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs when the host supports SVE-256bit.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> PreserveAll_DynamicFPRSVE = {
// v0 ~ v1 are used as temps.
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
@@ -176,20 +82,19 @@ namespace x64 {
namespace x32 {
// All but x19 and x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 10> SRA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
// PF/AF must be last.
REG_PF, REG_AF,
};
constexpr std::array<FEXCore::ARMEmitter::Register, 15> RA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
// All these callee saved
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
// Registers only available on 32-bit
// All these are caller saved (except for r19).
@@ -201,10 +106,11 @@ namespace x32 {
FEXCore::ARMEmitter::Reg::r19,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 7> RAPair = {{
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
@@ -221,106 +127,11 @@ namespace x32 {
};
// v8..v15 = (lower 64bits) Callee saved
constexpr std::array<FEXCore::ARMEmitter::VRegister, 22> RAFPR = {
// v0 ~ v1 are used as temps.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 20> RAFPR = {
// v0 ~ v3 are used as temps.
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
};
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 5> PreserveAll_SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8,
};
constexpr uint32_t PreserveAll_SRAMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRA) {
switch (Reg.Idx()) {
case 0:
case 1:
case 2:
case 3:
case 4:
case 5:
case 6:
case 7:
case 8:
case 16:
case 17:
Mask |= (1U << Reg.Idx());
break;
default: break;
}
}
return Mask;
}()
};
// Dynamic GPRs
constexpr std::array<FEXCore::ARMEmitter::Register, 3> PreserveAll_Dynamic = {
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r30
};
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
// None.
};
constexpr uint32_t PreserveAll_SRAFPRMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPR) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs
// - v0-v7
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
// v0 ~ v1 are temps
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
};
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
// This is /all/ of the SRA registers
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> PreserveAll_SRAFPRSVE = SRAFPR;
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPRSVE) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs when the host supports SVE-256bit.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 22> PreserveAll_DynamicFPRSVE = {
// v0 ~ v1 are used as temps.
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
@@ -336,8 +147,8 @@ namespace x32 {
}
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr, size_t size)
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
: Emitter(size ? (uint8_t*)FEXCore::Allocator::VirtualAlloc(size, true) : nullptr, size)
, EmitterCTX {ctx}
#ifdef VIXL_SIMULATOR
, Simulator {&SimDecoder}
@@ -369,7 +180,7 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
GeneralFPRegisters = x64::RAFPR;
}
else {
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
StaticRegisters = x32::SRA;
GeneralRegisters = x32::RA;
@@ -380,6 +191,13 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
}
}
Arm64Emitter::~Arm64Emitter() {
auto BufferSize = GetBufferSize();
if (BufferSize) {
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
}
}
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
int Segments = Is64Bit ? 4 : 2;
@@ -400,15 +218,6 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
Segments = 2;
}
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
movn(s, Reg.W(), (~Constant) & 0xFFFF);
if (NOPPad) {
nop(); nop(); nop();
}
return;
}
int RequiredMoveSegments{};
// Count the number of move segments
@@ -584,42 +393,10 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
}
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
#ifndef VIXL_SIMULATOR
if (EmitterCTX->HostFeatures.SupportsAFP) {
// Disable AFP features when spilling registers.
//
// Disable FPCR.NEP and FPCR.AH
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
// Also interacts with RPRES to change reciprocal/rsqrt precision from 8-bit mantissa to 12-bit.
//
// Additional interesting AFP bits:
// FIZ(0): Flush Inputs to Zero
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
(1U << 2) | // NEP
(1U << 1)); // AH
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
}
#endif
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
// is always static and almost certainly clobbered by the subsequent code.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
if (!StaticRegisterAllocation()) {
return;
}
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
GPRSpillMask &= ~PFAFSpillMask;
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i+1];
@@ -635,14 +412,6 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
}
}
// Now handle PF/AF
if (PFAFSpillMask) {
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
}
if (FPRs) {
if (EmitterCTX->HostFeatures.SupportsAVX) {
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
@@ -688,46 +457,6 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
}
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
[[maybe_unused]] bool FoundRegister{};
for (auto Reg : StaticRegisters) {
if (((1U << Reg.Idx()) & GPRFillMask)) {
TmpReg = Reg;
FoundRegister = true;
break;
}
}
LOGMAN_THROW_A_FMT(FoundRegister, "Didn't have an SRA register to use as a temporary while spilling!");
#ifndef VIXL_SIMULATOR
if (EmitterCTX->HostFeatures.SupportsAFP) {
// Enable AFP features when filling JIT state.
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
// Enable FPCR.NEP and FPCR.AH
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
//
// Additional interesting AFP bits:
// FIZ(0): Flush Inputs to Zero
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
(1U << 2) | // NEP
(1U << 1)); // AH
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
}
#endif
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
// is always static and was almost certainly clobbered.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
if (!StaticRegisterAllocation()) {
return;
}
@@ -738,11 +467,11 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
// since all that matters is we restore them on a fill.
// It's not a concern if they get trounced by something else.
if (EmitterCTX->HostFeatures.SupportsSVE) {
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
}
if (EmitterCTX->HostFeatures.SupportsAVX) {
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
const auto Reg = StaticFPRegisters[i];
@@ -755,6 +484,8 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
if (GPRFillMask && FPRFillMask == ~0U) {
// Optimize the common case where we can fill four registers per instruction.
// Use one of the filling static registers before we fill it.
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
// Load the sse offset in to the temporary register
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
@@ -785,11 +516,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
}
}
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
GPRFillMask &= ~PFAFMask;
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i+1];
@@ -804,115 +530,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
}
}
// Now handle PF/AF
if (PFAFFillMask) {
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
}
}
void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
if (SVERegs) {
size_t i = 0;
for (; i < (VRegs.size() % 4); i += 2) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
st2b(Reg1.Z(), Reg2.Z(), PRED_TMP_32B, TmpReg, 0);
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 2);
}
for (; i < VRegs.size(); i += 4) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
const auto Reg3 = VRegs[i + 2];
const auto Reg4 = VRegs[i + 3];
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
}
}
else {
size_t i = 0;
for (; i < (VRegs.size() % 4); i += 2) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), TmpReg, 32);
}
for (; i < VRegs.size(); i += 4) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
const auto Reg3 = VRegs[i + 2];
const auto Reg4 = VRegs[i + 3];
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
}
}
}
void Arm64Emitter::PushGeneralRegisters(FEXCore::ARMEmitter::Register TmpReg, std::span<const FEXCore::ARMEmitter::Register> Regs) {
size_t i = 0;
for (; i < (Regs.size() % 2); ++i) {
const auto Reg1 = Regs[i];
str<ARMEmitter::IndexType::POST>(Reg1.X(), TmpReg, 16);
}
for (; i < Regs.size(); i += 2) {
const auto Reg1 = Regs[i];
const auto Reg2 = Regs[i + 1];
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
}
}
void Arm64Emitter::PopVectorRegisters(bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
if (SVERegs) {
size_t i = 0;
for (; i < (VRegs.size() % 4); i += 2) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
ld2b(Reg1.Z(), Reg2.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 2);
}
for (; i < VRegs.size(); i += 4) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
const auto Reg3 = VRegs[i + 2];
const auto Reg4 = VRegs[i + 3];
ld4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
}
} else {
size_t i = 0;
for (; i < (VRegs.size() % 4); i += 2) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), ARMEmitter::Reg::rsp, 32);
}
for (; i < VRegs.size(); i += 4) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
const auto Reg3 = VRegs[i + 2];
const auto Reg4 = VRegs[i + 3];
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
}
}
}
void Arm64Emitter::PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Register> Regs) {
size_t i = 0;
for (; i < (Regs.size() % 2); ++i) {
const auto Reg1 = Regs[i];
ldr<ARMEmitter::IndexType::POST>(Reg1.X(), ARMEmitter::Reg::rsp, 16);
}
for (; i < Regs.size(); i += 2) {
const auto Reg1 = Regs[i];
const auto Reg2 = Regs[i + 1];
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
}
}
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
@@ -928,13 +545,31 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
// rsp capable move
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 2 == 0, "Needs to have multiple of 2 FPRs for RA");
if (CanUseSVE) {
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
const auto Reg1 = GeneralFPRegisters[i];
const auto Reg2 = GeneralFPRegisters[i + 1];
const auto Reg3 = GeneralFPRegisters[i + 2];
const auto Reg4 = GeneralFPRegisters[i + 3];
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
}
} else {
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
const auto Reg1 = GeneralFPRegisters[i];
const auto Reg2 = GeneralFPRegisters[i + 1];
const auto Reg3 = GeneralFPRegisters[i + 2];
const auto Reg4 = GeneralFPRegisters[i + 3];
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
}
}
// Push the vector registers
PushVectorRegisters(TmpReg, CanUseSVE, GeneralFPRegisters);
// Push the general registers.
PushGeneralRegisters(TmpReg, ConfiguredDynamicRegisterBase);
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
}
str(ARMEmitter::XReg::lr, TmpReg, 0);
}
@@ -942,107 +577,34 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
void Arm64Emitter::PopDynamicRegsAndLR() {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
// Pop vectors first
PopVectorRegisters(CanUseSVE, GeneralFPRegisters);
if (CanUseSVE) {
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
const auto Reg1 = GeneralFPRegisters[i];
const auto Reg2 = GeneralFPRegisters[i + 1];
const auto Reg3 = GeneralFPRegisters[i + 2];
const auto Reg4 = GeneralFPRegisters[i + 3];
ld4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
}
} else {
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
const auto Reg1 = GeneralFPRegisters[i];
const auto Reg2 = GeneralFPRegisters[i + 1];
const auto Reg3 = GeneralFPRegisters[i + 2];
const auto Reg4 = GeneralFPRegisters[i + 3];
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
}
}
// Pop GPRs second
PopGeneralRegisters(ConfiguredDynamicRegisterBase);
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
}
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
}
void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs) {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
: Core::CPUState::XMM_SSE_REG_SIZE;
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs{};
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs{};
uint32_t PreserveSRAMask{};
uint32_t PreserveSRAFPRMask{};
if (EmitterCTX->Config.Is64BitMode()) {
DynamicGPRs = x64::PreserveAll_Dynamic;
DynamicFPRs = x64::PreserveAll_DynamicFPR;
PreserveSRAMask = x64::PreserveAll_SRAMask;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRMask;
if (CanUseSVE) {
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
}
}
else {
DynamicGPRs = x32::PreserveAll_Dynamic;
DynamicFPRs = x32::PreserveAll_DynamicFPR;
PreserveSRAMask = x32::PreserveAll_SRAMask;
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRMask;
if (CanUseSVE) {
DynamicFPRs = x32::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRSVEMask;
}
}
const auto GPRSize = AlignUp(DynamicGPRs.size(), 2) * Core::CPUState::GPR_REG_SIZE;
const auto FPRSize = DynamicFPRs.size() * FPRRegSize;
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
// Spill the static registers.
SpillStaticRegs(TmpReg, true, PreserveSRAMask, PreserveSRAFPRMask);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
// rsp capable move
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
// Push the vector registers.
PushVectorRegisters(TmpReg, CanUseSVE, DynamicFPRs);
// Push the general registers.
PushGeneralRegisters(TmpReg, DynamicGPRs);
}
void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs{};
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs{};
uint32_t PreserveSRAMask{};
uint32_t PreserveSRAFPRMask{};
if (EmitterCTX->Config.Is64BitMode()) {
DynamicGPRs = x64::PreserveAll_Dynamic;
DynamicFPRs = x64::PreserveAll_DynamicFPR;
PreserveSRAMask = x64::PreserveAll_SRAMask;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRMask;
if (CanUseSVE) {
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
}
}
else {
DynamicGPRs = x32::PreserveAll_Dynamic;
DynamicFPRs = x32::PreserveAll_DynamicFPR;
PreserveSRAMask = x32::PreserveAll_SRAMask;
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRMask;
if (CanUseSVE) {
DynamicFPRs = x32::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRSVEMask;
}
}
// Fill the static registers.
FillStaticRegs(true, PreserveSRAMask, PreserveSRAFPRMask);
// Pop the vector registers.
PopVectorRegisters(CanUseSVE, DynamicFPRs);
// Pop the general registers.
PopGeneralRegisters(DynamicGPRs);
}
void Arm64Emitter::Align16B() {
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
@@ -1,10 +1,10 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "FEXCore/Utils/EnumUtils.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/ObjectCache/Relocations.h"
#include <aarch64/assembler-aarch64.h>
@@ -28,10 +28,6 @@
#include <utility>
#include <span>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::CPU {
// Contains the address to the currently available CPU state
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
@@ -46,6 +42,8 @@ constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x3;
// Vector temporaries
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v0;
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
constexpr auto VTMP3 = FEXCore::ARMEmitter::VReg::v2;
constexpr auto VTMP4 = FEXCore::ARMEmitter::VReg::v3;
// Predicate register temporaries (used when AVX support is enabled)
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
@@ -53,15 +51,12 @@ constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
protected:
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr = nullptr, size_t size = 0);
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size);
~Arm64Emitter();
FEXCore::Context::ContextImpl *EmitterCTX;
vixl::aarch64::CPU CPU;
@@ -104,52 +99,12 @@ protected:
// We can't guarantee only the lower 64bits are used so flush everything
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
// Generic push and pop vector registers.
void PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs);
void PushGeneralRegisters(FEXCore::ARMEmitter::Register TmpReg, std::span<const FEXCore::ARMEmitter::Register> Regs);
void PopVectorRegisters(bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs);
void PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Register> Regs);
void PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg);
void PopDynamicRegsAndLR();
void PushCalleeSavedRegisters();
void PopCalleeSavedRegisters();
// Spills and fills SRA/Dynamic registers that are required for Arm64 `preserve_all` ABI.
// This ABI changes most registers to be callee saved.
// Caller Saved:
// - X0-X8, X16-X18.
// - v0-v7
// - For 256-bit SVE hosts: top 128-bits of v8-v31
//
// Callee Saved:
// - X9-X15, X19-X31
// - Low 128-bits of v8-v31
void SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true);
void FillForPreserveAllABICall(bool FPRs = true);
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
if (SupportsPreserveAllABI) {
SpillForPreserveAllABICall(TmpReg, FPRs);
}
else {
SpillStaticRegs(TmpReg, FPRs);
PushDynamicRegsAndLR(TmpReg);
}
}
void FillForABICall(bool SupportsPreserveAllABI, bool FPRs = true) {
if (SupportsPreserveAllABI) {
FillForPreserveAllABICall(FPRs);
}
else {
PopDynamicRegsAndLR();
FillStaticRegs(FPRs);
}
}
void Align16B();
#ifdef VIXL_SIMULATOR
@@ -216,15 +171,7 @@ protected:
// Call type
dc32(vixl::aarch64::kCallRuntime);
}
#else
template<typename R, typename... P>
void GenerateRuntimeCall(R (*Function)(P...)) {
// Explicitly doing nothing.
}
template<typename R, typename... P>
void GenerateIndirectRuntimeCall(ARMEmitter::Register Reg) {
// Explicitly doing nothing.
}
#endif
#ifdef VIXL_SIMULATOR
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* ALU instruction emitters.
*
* Almost all of these operations have `ARMEmitter::Size` as their first argument.
@@ -766,21 +765,6 @@ public:
EvaluateIntoFlags(Op, 1, rn);
}
void cfinv() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0001'1111;
dc32(Op);
}
void axflag() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0101'1111;
dc32(Op);
}
void xaflag() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0011'1111;
dc32(Op);
}
// Conditional compare - register
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0011'1010'010 << 21;
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* ASIMD instruction emitters.
*
* This contains emitters for vector operations explicitly.
@@ -60,7 +59,7 @@ public:
}
void sha256su1(FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn, FEXCore::ARMEmitter::VRegister rm) {
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
Crypto3RegSHA(Op, 0b110, rd, rn, rm);
Crypto3RegSHA(Op, 0b100, rd, rn, rm);
}
// Cryptographic two-register SHA
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* Branch instruction emitters.
*
* Most of these instructions will use `BackwardLabel`, `ForwardLabel`, or `BiDirectionLabel` to determine where a branch targets.
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <cstddef>
#include <cstdint>
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Interface/Core/ArchHelpers/CodeEmitter/Buffer.h"
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* Load-store instruction emitters
*
* For GPR load-stores that take a `Size` argument as their first argument can be 32-bit or 64-bit.
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/EnumUtils.h>
@@ -21,11 +20,12 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const Register&, const Register&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr WRegister W() const;
constexpr XRegister X() const;
WRegister W() const;
XRegister X() const;
private:
uint32_t Index;
@@ -45,15 +45,16 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const WRegister&, const WRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr operator Register() const {
operator Register() const {
return Register(Index);
}
constexpr XRegister X() const;
constexpr Register R() const;
XRegister X() const;
Register R() const;
private:
uint32_t Index;
@@ -73,15 +74,16 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const XRegister&, const XRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr operator Register() const {
operator Register() const {
return Register(Index);
}
constexpr WRegister W() const;
constexpr Register R() const;
WRegister W() const;
Register R() const;
private:
uint32_t Index;
@@ -90,27 +92,27 @@ namespace FEXCore::ARMEmitter {
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
inline constexpr WRegister Register::W() const {
inline WRegister Register::W() const {
return WRegister{Index};
}
inline constexpr XRegister Register::X() const {
inline XRegister Register::X() const {
return XRegister{Index};
}
inline constexpr XRegister WRegister::X() const {
inline XRegister WRegister::X() const {
return XRegister{Index};
}
inline constexpr Register WRegister::R() const {
inline Register WRegister::R() const {
return *this;
}
inline constexpr WRegister XRegister::W() const {
inline WRegister XRegister::W() const {
return WRegister{Index};
}
inline constexpr Register XRegister::R() const {
inline Register XRegister::R() const {
return *this;
}
@@ -257,6 +259,7 @@ namespace FEXCore::ARMEmitter {
class QRegister;
class ZRegister;
/* Unsized ASIMD register class
* This class doesn't imply a size when used, nor implies Vector or Scalar.
* It does imply that this instruction isn't using the register for SVE.
@@ -269,16 +272,16 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const VRegister&, const VRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr BRegister B() const;
constexpr HRegister H() const;
constexpr SRegister S() const;
constexpr DRegister D() const;
constexpr QRegister Q() const;
constexpr ZRegister Z() const;
BRegister B() const;
HRegister H() const;
SRegister S() const;
DRegister D() const;
QRegister Q() const;
ZRegister Z() const;
private:
uint32_t Index;
@@ -298,19 +301,20 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const BRegister&, const BRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr operator VRegister() const {
operator VRegister () const {
return VRegister(Index);
}
constexpr BRegister V() const;
constexpr HRegister H() const;
constexpr SRegister S() const;
constexpr DRegister D() const;
constexpr QRegister Q() const;
constexpr ZRegister Z() const;
BRegister V() const;
HRegister H() const;
SRegister S() const;
DRegister D() const;
QRegister Q() const;
ZRegister Z() const;
private:
uint32_t Index;
@@ -330,19 +334,20 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const HRegister&, const HRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr operator VRegister() const {
operator VRegister() const {
return VRegister(Index);
}
constexpr HRegister V() const;
constexpr BRegister B() const;
constexpr SRegister S() const;
constexpr DRegister D() const;
constexpr QRegister Q() const;
constexpr ZRegister Z() const;
HRegister V() const;
BRegister B() const;
SRegister S() const;
DRegister D() const;
QRegister Q() const;
ZRegister Z() const;
private:
uint32_t Index;
@@ -362,19 +367,20 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const SRegister&, const SRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr operator VRegister() const {
operator VRegister() const {
return VRegister(Index);
}
constexpr SRegister V() const;
constexpr BRegister B() const;
constexpr HRegister H() const;
constexpr DRegister D() const;
constexpr QRegister Q() const;
constexpr ZRegister Z() const;
SRegister V() const;
BRegister B() const;
HRegister H() const;
DRegister D() const;
QRegister Q() const;
ZRegister Z() const;
private:
uint32_t Index;
@@ -395,19 +401,20 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const DRegister&, const DRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr operator VRegister() const {
operator VRegister() const {
return VRegister(Index);
}
constexpr DRegister V() const;
constexpr BRegister B() const;
constexpr HRegister H() const;
constexpr SRegister S() const;
constexpr QRegister Q() const;
constexpr ZRegister Z() const;
DRegister V() const;
BRegister B() const;
HRegister H() const;
SRegister S() const;
QRegister Q() const;
ZRegister Z() const;
private:
uint32_t Index;
@@ -428,19 +435,20 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const QRegister&, const QRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr operator VRegister() const {
operator VRegister () const {
return VRegister(Index);
}
constexpr QRegister V() const;
constexpr BRegister B() const;
constexpr HRegister H() const;
constexpr SRegister S() const;
constexpr DRegister D() const;
constexpr ZRegister Z() const;
QRegister V() const;
BRegister B() const;
HRegister H() const;
SRegister S() const;
DRegister D() const;
ZRegister Z() const;
private:
uint32_t Index;
@@ -460,16 +468,16 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const ZRegister&, const ZRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr VRegister V() const;
constexpr BRegister B() const;
constexpr HRegister H() const;
constexpr SRegister S() const;
constexpr DRegister D() const;
constexpr QRegister Q() const;
VRegister V() const;
BRegister B() const;
HRegister H() const;
SRegister S() const;
DRegister D() const;
QRegister Q() const;
private:
uint32_t Index;
@@ -479,142 +487,142 @@ namespace FEXCore::ARMEmitter {
static_assert(std::is_standard_layout_v<ZRegister>, "Needs to be standard");
// VRegister
inline constexpr BRegister VRegister::B() const {
inline BRegister VRegister::B() const {
return BRegister{Index};
}
inline constexpr HRegister VRegister::H() const {
inline HRegister VRegister::H() const {
return HRegister{Index};
}
inline constexpr SRegister VRegister::S() const {
inline SRegister VRegister::S() const {
return SRegister{Index};
}
inline constexpr DRegister VRegister::D() const {
inline DRegister VRegister::D() const {
return DRegister{Index};
}
inline constexpr QRegister VRegister::Q() const {
inline QRegister VRegister::Q() const {
return QRegister{Index};
}
inline constexpr ZRegister VRegister::Z() const {
inline ZRegister VRegister::Z() const {
return ZRegister{Index};
}
// BRegister
inline constexpr BRegister BRegister::V() const {
inline BRegister BRegister::V() const {
return *this;
}
inline constexpr HRegister BRegister::H() const {
inline HRegister BRegister::H() const {
return HRegister{Index};
}
inline constexpr SRegister BRegister::S() const {
inline SRegister BRegister::S() const {
return SRegister{Index};
}
inline constexpr DRegister BRegister::D() const {
inline DRegister BRegister::D() const {
return DRegister{Index};
}
inline constexpr QRegister BRegister::Q() const {
inline QRegister BRegister::Q() const {
return QRegister{Index};
}
inline constexpr ZRegister BRegister::Z() const {
inline ZRegister BRegister::Z() const {
return ZRegister{Index};
}
// HRegister
inline constexpr HRegister HRegister::V() const {
inline HRegister HRegister::V() const {
return *this;
}
inline constexpr BRegister HRegister::B() const {
inline BRegister HRegister::B() const {
return BRegister{Index};
}
inline constexpr SRegister HRegister::S() const {
inline SRegister HRegister::S() const {
return SRegister{Index};
}
inline constexpr DRegister HRegister::D() const {
inline DRegister HRegister::D() const {
return DRegister{Index};
}
inline constexpr QRegister HRegister::Q() const {
inline QRegister HRegister::Q() const {
return QRegister{Index};
}
inline constexpr ZRegister HRegister::Z() const {
inline ZRegister HRegister::Z() const {
return ZRegister{Index};
}
// SRegister
inline constexpr SRegister SRegister::V() const {
inline SRegister SRegister::V() const {
return *this;
}
inline constexpr BRegister SRegister::B() const {
inline BRegister SRegister::B() const {
return BRegister{Index};
}
inline constexpr HRegister SRegister::H() const {
inline HRegister SRegister::H() const {
return HRegister{Index};
}
inline constexpr DRegister SRegister::D() const {
inline DRegister SRegister::D() const {
return DRegister{Index};
}
inline constexpr QRegister SRegister::Q() const {
inline QRegister SRegister::Q() const {
return QRegister{Index};
}
inline constexpr ZRegister SRegister::Z() const {
inline ZRegister SRegister::Z() const {
return ZRegister{Index};
}
// DRegister
inline constexpr DRegister DRegister::V() const {
inline DRegister DRegister::V() const {
return DRegister{Index};
}
inline constexpr BRegister DRegister::B() const {
inline BRegister DRegister::B() const {
return BRegister{Index};
}
inline constexpr HRegister DRegister::H() const {
inline HRegister DRegister::H() const {
return HRegister{Index};
}
inline constexpr SRegister DRegister::S() const {
inline SRegister DRegister::S() const {
return SRegister{Index};
}
inline constexpr QRegister DRegister::Q() const {
inline QRegister DRegister::Q() const {
return QRegister{Index};
}
inline constexpr ZRegister DRegister::Z() const {
inline ZRegister DRegister::Z() const {
return ZRegister{Index};
}
// QRegister
inline constexpr QRegister QRegister::V() const {
inline QRegister QRegister::V() const {
return *this;
}
inline constexpr BRegister QRegister::B() const {
inline BRegister QRegister::B() const {
return BRegister{Index};
}
inline constexpr HRegister QRegister::H() const {
inline HRegister QRegister::H() const {
return HRegister{Index};
}
inline constexpr SRegister QRegister::S() const {
inline SRegister QRegister::S() const {
return SRegister{Index};
}
inline constexpr DRegister QRegister::D() const {
inline DRegister QRegister::D() const {
return DRegister{Index};
}
inline constexpr ZRegister QRegister::Z() const {
inline ZRegister QRegister::Z() const {
return ZRegister{Index};
}
// ZRegister
inline constexpr VRegister ZRegister::V() const {
inline VRegister ZRegister::V() const {
return VRegister(Index);
}
inline constexpr BRegister ZRegister::B() const {
inline BRegister ZRegister::B() const {
return BRegister(Index);
}
inline constexpr HRegister ZRegister::H() const {
inline HRegister ZRegister::H() const {
return HRegister(Index);
}
inline constexpr SRegister ZRegister::S() const {
inline SRegister ZRegister::S() const {
return SRegister(Index);
}
inline constexpr DRegister ZRegister::D() const {
inline DRegister ZRegister::D() const {
return DRegister(Index);
}
inline constexpr QRegister ZRegister::Q() const {
inline QRegister ZRegister::Q() const {
return QRegister(Index);
}
@@ -871,28 +879,36 @@ namespace FEXCore::ARMEmitter {
}
// Zero-cost FPR->GPR
inline constexpr Register ToReg(HRegister Reg) {
return Register(Reg.Idx());
inline
Register ToReg(HRegister Reg) {
return static_cast<Register>(Reg.Idx());
}
inline constexpr Register ToReg(SRegister Reg) {
return Register(Reg.Idx());
inline
Register ToReg(SRegister Reg) {
return static_cast<Register>(Reg.Idx());
}
inline constexpr Register ToReg(DRegister Reg) {
return Register(Reg.Idx());
inline
Register ToReg(DRegister Reg) {
return static_cast<Register>(Reg.Idx());
}
inline constexpr Register ToReg(VRegister Reg) {
return Register(Reg.Idx());
inline
Register ToReg(VRegister Reg) {
return static_cast<Register>(Reg.Idx());
}
// Zero-cost GPR->FPR
inline constexpr VRegister ToVReg(Register Reg) {
return VRegister(Reg.Idx());
inline
VRegister ToVReg(Register Reg) {
return static_cast<VRegister>(Reg.Idx());
}
inline constexpr VRegister ToVReg(XRegister Reg) {
return VRegister(Reg.Idx());
inline
VRegister ToVReg(XRegister Reg) {
return static_cast<VRegister>(Reg.Idx());
}
inline constexpr VRegister ToVReg(WRegister Reg) {
return VRegister(Reg.Idx());
inline
VRegister ToVReg(WRegister Reg) {
return static_cast<VRegister>(Reg.Idx());
}
class PRegisterZero;
@@ -909,12 +925,12 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const PRegister&, const PRegister&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr PRegisterZero Zeroing() const;
constexpr PRegisterMerge Merging() const;
PRegisterZero Zeroing() const;
PRegisterMerge Merging() const;
private:
uint32_t Index;
@@ -932,17 +948,14 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const PRegisterZero&, const PRegisterZero&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr operator PRegister() const {
return PRegister(Index);
}
constexpr PRegister P() const {
return PRegister(Index);
}
constexpr PRegisterMerge Merging() const;
operator PRegister() const;
PRegister P() const;
PRegisterMerge Merging() const;
private:
uint32_t Index;
@@ -960,17 +973,14 @@ namespace FEXCore::ARMEmitter {
friend constexpr auto operator<=>(const PRegisterMerge&, const PRegisterMerge&) = default;
constexpr uint32_t Idx() const {
uint32_t Idx() const {
return Index;
}
constexpr operator PRegister() const {
return PRegister(Index);
}
constexpr PRegister P() const {
return PRegister(Index);
}
constexpr PRegisterZero Zeroing() const;
operator PRegister() const;
PRegister P() const;
PRegisterZero Zeroing() const;
private:
uint32_t Index;
@@ -979,21 +989,39 @@ namespace FEXCore::ARMEmitter {
static_assert(std::is_trivial_v<PRegisterZero>, "Needs to be trivial");
static_assert(std::is_standard_layout_v<PRegisterZero>, "Needs to be standard");
// PRegister
inline constexpr PRegisterZero PRegister::Zeroing() const {
inline PRegisterZero PRegister::Zeroing() const {
return PRegisterZero(Idx());
}
inline constexpr PRegisterMerge PRegister::Merging() const {
inline PRegisterMerge PRegister::Merging() const {
return PRegisterMerge(Idx());
}
// PRegisterZero
inline constexpr PRegisterMerge PRegisterZero::Merging() const {
inline PRegisterZero::operator PRegister() const {
return PRegister(Index);
}
inline PRegister PRegisterZero::P() const {
return PRegister(Idx());
}
inline PRegisterMerge PRegisterZero::Merging() const {
return PRegisterMerge(Idx());
}
// PRegisterMerge
inline constexpr PRegisterZero PRegisterMerge::Zeroing() const {
inline PRegisterMerge::operator PRegister() const {
return PRegisterZero(Index);
}
inline PRegister PRegisterMerge::P() const {
return PRegister(Idx());
}
inline PRegisterZero PRegisterMerge::Zeroing() const {
return PRegisterZero(Idx());
}
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* SVE instruction emitters
* These contain instruction emitters for AArch64 SVE and SVE2 operations.
*
@@ -1384,10 +1383,12 @@ public:
}
// SVE predicate initialize
void ptrue(SubRegSize size, PRegister pd, PredicatePattern pattern) {
template <SubRegSize size>
void ptrue(PRegister pd, PredicatePattern pattern) {
SVEPredicateMisc(0b1000, 0b10000, FEXCore::ToUnderlying(pattern), size, pd);
}
void ptrues(SubRegSize size, PRegister pd, PredicatePattern pattern) {
template <SubRegSize size>
void ptrues(PRegister pd, PredicatePattern pattern) {
SVEPredicateMisc(0b1001, 0b10000, FEXCore::ToUnderlying(pattern), size, pd);
}
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* Scalar instruction emitters.
*
* These contain instruction emitters for scalar ASIMD operations explicitly.
@@ -798,52 +797,6 @@ public:
// XXX:
//
// Floating-point data-processing (1 source)
void fmov(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b000000, rd, rn);
}
void fabs(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b000001, rd, rn);
}
void fneg(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b000010, rd, rn);
}
void fsqrt(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b000011, rd, rn);
}
void frintn(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b001000, rd, rn);
}
void frintp(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b001001, rd, rn);
}
void frintm(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b001010, rd, rn);
}
void frintz(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b001011, rd, rn);
}
void frinta(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b001100, rd, rn);
}
void frintx(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b001110, rd, rn);
}
void frinti(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b001111, rd, rn);
}
void frint32z(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b010000, rd, rn);
}
void frint32x(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b010001, rd, rn);
}
void frint64z(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b010010, rd, rn);
}
void frint64x(ScalarRegSize size, VRegister rd, VRegister rn) {
Float1Source(size, 0, 0, 0b010011, rd, rn);
}
void fmov(SRegister rd, SRegister rn) {
Float1Source(0, 0, 0b00, 0b000000, rd.V(), rn.V());
}
@@ -1111,34 +1064,6 @@ public:
}
// Floating-point data-processing (2 source)
void fmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
Float2Source(size, 0, 0, 0b0000, rd, rn, rm);
}
void fdiv(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
Float2Source(size, 0, 0, 0b0001, rd, rn, rm);
}
void fadd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
Float2Source(size, 0, 0, 0b0010, rd, rn, rm);
}
void fsub(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
Float2Source(size, 0, 0, 0b0011, rd, rn, rm);
}
void fmax(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
Float2Source(size, 0, 0, 0b0100, rd, rn, rm);
}
void fmin(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
Float2Source(size, 0, 0, 0b0101, rd, rn, rm);
}
void fmaxnm(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
Float2Source(size, 0, 0, 0b0110, rd, rn, rm);
}
void fminnm(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
Float2Source(size, 0, 0, 0b0111, rd, rn, rm);
}
void fnmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
Float2Source(size, 0, 0, 0b1000, rd, rn, rm);
}
void fmul(SRegister rd, SRegister rn, SRegister rm) {
Float2Source(0, 0, 0b00, 0b0000, rd.V(), rn.V(), rm.V());
}
@@ -1224,16 +1149,6 @@ public:
}
// Floating-point conditional select
void fcsel(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
const uint32_t ConvertedSize =
size == ScalarRegSize::i64Bit ? 0b01 :
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
FloatConditionalSelect(0, 0, ConvertedSize, rd, rn, rm, Cond);
}
void fcsel(SRegister rd, SRegister rn, SRegister rm, Condition Cond) {
FloatConditionalSelect(0, 0, 0b00, rd.V(), rn.V(), rm.V(), Cond);
}
@@ -1389,16 +1304,6 @@ private:
dc32(Instr);
}
void Float1Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn) {
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
const uint32_t ConvertedSize =
size == ScalarRegSize::i64Bit ? 0b01 :
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
Float1Source(M, S, ConvertedSize, opcode, rd, rn);
}
// Floating-point compare
void FloatCompare(uint32_t M, uint32_t S, uint32_t ftype, uint32_t op, uint32_t opcode2, VRegister rn, VRegister rm) {
uint32_t Instr = 0b0001'1110'0010'0000'0010'0000'0000'0000;
@@ -1431,7 +1336,6 @@ private:
dc32(Instr);
}
// Floating-point data-processing (2 source)
void Float2Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
uint32_t Instr = 0b0001'1110'0010'0000'0000'1000'0000'0000;
@@ -1446,16 +1350,6 @@ private:
dc32(Instr);
}
void Float2Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
const uint32_t ConvertedSize =
size == ScalarRegSize::i64Bit ? 0b01 :
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
Float2Source(M, S, ConvertedSize, opcode, rd, rn, rm);
}
// Floating-point conditional select
void FloatConditionalSelect(uint32_t M, uint32_t S, uint32_t ptype, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
uint32_t Instr = 0b0001'1110'0010'0000'0000'1100'0000'0000;
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* System instruction emitters.
*
* This is mostly a mashup of various instruction types.
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/BlockSamplingData.h"
#include <FEXCore/Utils/LogManager.h>
#include <cstring>
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <cstdint>
Loaded 100 of 747 files, more files were not shown because too many files have changed in this diff. Show more