Compare commits

..
2 Commits
879 changed files with 69363 additions and 123725 deletions

No files matched your search

@@ -37,6 +37,7 @@ If applicable, add screenshots and video to help explain your problem.
**Additional context**
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
- Is this a Vulkan game: [Yes/No/Unknown]
- If Yes, What is your Vulkan driver:
+63 -50
View File
@@ -13,14 +13,15 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
FEX_ENABLEAVX: 1
jobs:
build_plus_test:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARMv8.0], [self-hosted, ARMv8.2], [self-hosted, ARMv8.4]]
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2], [self-hosted, ARMv8.4]]
fail-fast: false
steps:
@@ -64,7 +65,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -77,6 +78,54 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: gvisor tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gvisor_tests
- name: GVisor Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GVisor.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -101,6 +150,17 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
- name: Struct verifier tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target struct_verifier
- name: Struct verifier Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
- name: APITest tests
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -184,53 +244,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkResults.log || true
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: gvisor tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gvisor_tests
- name: GVisor Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GVisor.log || true
- name: Struct verifier tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target struct_verifier
- name: Struct verifier Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
+41 -27
View File
@@ -20,14 +20,16 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
FEX_ENABLEAVX: 1
jobs:
glibc_fault_test:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARM64]]
# Run on an x86 device and any ARM runner.
arch: [[self-hosted, x64], [self-hosted, ARM64]]
fail-fast: false
steps:
@@ -71,7 +73,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -84,6 +86,42 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -141,30 +179,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
-107
View File
@@ -1,107 +0,0 @@
name: Hostrunner tests
on:
push:
branches:
- main
pull_request:
branches:
- main
env:
# Customize the CMake build type here (Release, Debug, RelWithDebInfo, etc.)
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_ENABLEAVX: 1
jobs:
hostrunner_tests:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, x64]]
fail-fast: false
steps:
- uses: actions/checkout@v3
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: 'actions/upload-artifact@v3'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
+3 -41
View File
@@ -16,11 +16,11 @@ env:
FEX_ENABLEAVX: 1
jobs:
instcountci_tests:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, x64], [self-hosted, ARM64]]
arch: [[self-hosted, ARM64]]
fail-fast: false
steps:
@@ -56,16 +56,6 @@ jobs:
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Set vixl_sim x86
if: matrix.arch[1] == 'x64'
run: |
echo "VIXL_SIM_ENABLED=True" >> $GITHUB_ENV
- name: Set vixl_sim Arm64
if: matrix.arch[1] == 'ARM64'
run: |
echo "VIXL_SIM_ENABLED=False" >> $GITHUB_ENV
- name: Configure CMake
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
@@ -74,7 +64,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=False -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -96,25 +86,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_InstCountCI.log || true
- name: Update local repo instcount
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: cmake --build . --config $BUILD_TYPE --target instcountci_update_tests
- name: Get instcountCI diff
if: ${{ always() }}
shell: bash
working-directory: ${{github.workspace}}/
run: git diff --output=${{runner.workspace}}/build/InstCountCI.diff
- name: Check if InstCountCI Diff exists
if: ${{ always() }}
shell: bash
working-directory: ${{github.workspace}}/
# Check if the file is empty
run: sh -c "! test -s ${{runner.workspace}}/build/InstCountCI.diff"
- name: Truncate test results
if: ${{ always() }}
shell: bash
@@ -136,12 +107,3 @@ jobs:
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
- name: Upload results InstCountCI
if: ${{ always() }}
uses: 'actions/upload-artifact@v3'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}-instcountci
path: ${{runner.workspace}}/build/InstCountCI.diff
retention-days: 3
+3 -3
View File
@@ -13,11 +13,11 @@ env:
FEX_ENABLEAVX: 1
jobs:
mingw_build:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARM64, mingw]]
arch: [[self-hosted, x64, mingw], [self-hosted, ARM64, mingw]]
fail-fast: false
steps:
@@ -74,7 +74,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=False -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
+13 -1
View File
@@ -16,7 +16,7 @@ env:
FEX_ENABLEAVX: 1
jobs:
vixl_simulator:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
@@ -100,6 +100,18 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM128bit.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
+6 -6
View File
@@ -15,19 +15,19 @@
path = External/tiny-json
url = https://github.com/Sonicadvance1/tiny-json.git
[submodule "External/xbyak"]
shallow = true
shallow = true
path = External/xbyak
url = https://github.com/herumi/xbyak.git
url = https://github.com/FEX-Emu/xbyak.git
[submodule "External/fex-posixtest-bins"]
shallow = true
shallow = true
path = External/fex-posixtest-bins
url = https://github.com/FEX-Emu/fex-posixtest-bins.git
[submodule "External/fex-gvisor-tests-bins"]
shallow = true
shallow = true
path = External/fex-gvisor-tests-bins
url = https://github.com/FEX-Emu/fex-gvisor-tests-bins.git
[submodule "External/fex-gcc-target-tests-bins"]
shallow = true
shallow = true
path = External/fex-gcc-target-tests-bins
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
[submodule "External/jemalloc"]
@@ -41,7 +41,7 @@
url = https://github.com/FEX-Emu/drm-headers.git
[submodule "External/xxhash"]
path = External/xxhash
url = https://github.com/Cyan4973/xxHash.git
url = https://github.com/FEX-Emu/xxHash.git
[submodule "External/Catch2"]
path = External/Catch2
url = https://github.com/catchorg/Catch2.git
-5
View File
@@ -1,5 +0,0 @@
{
"ThunksDB": {
"fex_thunk_test": 1
}
}
+23 -17
View File
@@ -25,6 +25,7 @@ option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enables jemalloc glibc allocator" TRUE)
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
@@ -96,6 +97,11 @@ if (ENABLE_GDB_SYMBOLS)
endif()
if (ENABLE_INTERPRETER)
message(STATUS "Interpreter enabled")
add_definitions(-DINTERPRETER_ENABLED=1)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
@@ -112,6 +118,14 @@ else()
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
message(FATAL_ERROR
" Be warned: FEX isn't optimized for x86_64 hosts!\n"
" Support for x86_64 hosts is only for debugging and convenience!\n"
" Don't expect amazing performance or optimal code generation!\n"
" Pass -DENABLE_X86_HOST_DEBUG=True to bypass this message!")
endif()
set(_M_X86_64 1)
add_definitions(-D_M_X86_64=1)
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
@@ -122,11 +136,6 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
add_definitions(-D_M_ARM_64=1)
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
set(_M_ARM_64EC 1)
add_definitions(-D_M_ARM_64EC=1)
endif()
if (ENABLE_CCACHE)
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
@@ -237,10 +246,8 @@ endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
set(XXHASH_BUNDLED_MODE TRUE)
set(XXHASH_BUILD_XXHSUM FALSE)
set(BUILD_SHARED_LIBS OFF)
add_subdirectory(External/xxhash/cmake_unofficial/)
add_subdirectory(External/xxhash/)
include_directories(External/xxhash/)
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
@@ -300,11 +307,10 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
endif()
endif()
set(FEX_TUNE_COMPILE_FLAGS)
if (NOT TUNE_ARCH STREQUAL "generic")
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
if(COMPILER_SUPPORTS_ARCH_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH}")
add_compile_options("-march=${TUNE_ARCH}")
else()
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
endif()
@@ -317,7 +323,7 @@ if (TUNE_CPU STREQUAL "native")
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=native")
add_compile_options("-mcpu=native")
endif()
else()
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
@@ -331,19 +337,19 @@ if (TUNE_CPU STREQUAL "native")
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${AARCH64_CPU}")
add_compile_options("-mcpu=${AARCH64_CPU}")
endif()
endif()
else()
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
if(COMPILER_SUPPORTS_MARCH_NATIVE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=native")
add_compile_options("-march=native")
endif()
endif()
else()
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${TUNE_CPU}")
add_compile_options("-mcpu=${TUNE_CPU}")
else()
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
endif()
@@ -461,10 +467,10 @@ if (BUILD_THUNKS)
CMAKE_ARGS
"-DBITNESS=64"
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DBUILD_FEX_LINUX_TESTS=${BUILD_FEX_LINUX_TESTS}"
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_64_TOOLCHAIN_FILE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
INSTALL_COMMAND ""
@@ -479,10 +485,10 @@ if (BUILD_THUNKS)
CMAKE_ARGS
"-DBITNESS=32"
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DBUILD_FEX_LINUX_TESTS=${BUILD_FEX_LINUX_TESTS}"
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_32_TOOLCHAIN_FILE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
INSTALL_COMMAND ""
+5
View File
@@ -0,0 +1,5 @@
{
"Config": {
"Env": "STEAM_GAME_LAUNCH_SHELL=@CMAKE_INSTALL_PREFIX@/bin/FEXBash"
}
}
+5
View File
@@ -0,0 +1,5 @@
{
"Config": {
"AdditionalArguments": "--no-sandbox"
}
}
-6
View File
@@ -144,12 +144,6 @@
"@PREFIX_LIB@/libasound.so.2.0.0"
]
},
"fex_thunk_test": {
"Library": "libfex_thunk_test-guest.so",
"Overlay": [
"@PREFIX_LIB@/libfex_thunk_test.so"
]
},
"Xrender": {
"Library": "libXrender-guest.so",
"Overlay": [
+4 -5
View File
@@ -3,12 +3,11 @@ FROM ubuntu:20.04 as builder
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
clang-10 llvm-10 nasm ninja-build pkg-config \
libcap-dev libglfw3-dev libepoxy-dev python3-dev libsdl2-dev \
python3 linux-headers-generic \
git
clang-10 llvm-10 nasm ninja-build \
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
python3 linux-headers-generic
RUN git clone --recurse-submodules https://github.com/FEX-Emu/FEX.git
COPY . /opt/FEX
CMD [ "mkdir /opt/FEX/build" ]
+13
View File
@@ -0,0 +1,13 @@
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
Version 2, December 2004
Copyright (C) 2018 Ryan Houdek <Sonicadvance1@gmail.com>
Everyone is permitted to copy and distribute verbatim or modified
copies of this license document, and changing it is allowed as long
as the name is changed.
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION
0. You just DO WHAT THE FUCK YOU WANT TO.
+1 -1
+1 -1
+1 -1
+1 -1
+9 -22
View File
@@ -13,6 +13,15 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
set(_M_ARM_64 1)
endif()
if (ENABLE_VIXL_SIMULATOR)
# If the vixl simulator is enabled then we are using the ARM64 JIT
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" FALSE)
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" TRUE)
else()
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" ${_M_X86_64})
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" ${_M_ARM_64})
endif()
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
@@ -24,28 +33,6 @@ set(CMAKE_INCLUDE_CURRENT_DIR ON)
include(CheckCXXCompilerFlag)
include(CheckIncludeFileCXX)
include(CheckCXXSourceCompiles)
set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
check_cxx_source_compiles(
"
__attribute__((preserve_all))
int Testy(int a, int b, int c, int d, int e, int f) {
return a + b + c + d + e + f;
}
int main() {
return Testy(0, 1, 2, 3, 4, 5);
}"
HAS_CLANG_PRESERVE_ALL)
unset(CMAKE_REQUIRED_FLAGS)
if (HAS_CLANG_PRESERVE_ALL)
if (MINGW_BUILD)
message(STATUS "Ignoring broken clang::preserve_all support")
set(HAS_CLANG_PRESERVE_ALL FALSE)
else()
message(STATUS "Has clang::preserve_all")
endif()
endif ()
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
# Useful to have for freestanding libFEXCore
-21
View File
@@ -441,24 +441,6 @@ def print_parse_envloader_options(options):
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_jsonloader_options(options):
output_argloader.write("#ifdef JSONLOADER\n")
output_argloader.write("#undef JSONLOADER\n")
output_argloader.write("if (false) {}\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
value_type = op_vals["Type"]
if (value_type == "strenum"):
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
output_argloader.write("Set(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
output_argloader.write("}\n")
output_argloader.write("else {{\n".format(op_key))
output_argloader.write("Set(KeyOption, ConfigString);\n")
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_enum_options(options):
output_argloader.write("#ifdef ENUMDEFINES\n")
output_argloader.write("#undef ENUMDEFINES\n")
@@ -574,9 +556,6 @@ print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
# Generate json loader code
print_parse_jsonloader_options(options);
# Generate enum variable options
print_parse_enum_options(options);
+10 -73
View File
@@ -46,15 +46,11 @@ class OpDefinition:
NumElements: str
OpClass: str
HasSideEffects: bool
ImplicitFlagClobber: bool
RAOverride: int
SwitchGen: bool
ArgPrinter: bool
SSAArgNum: int
NonSSAArgNum: int
DynamicDispatch: bool
JITDispatch: bool
JITDispatchOverride: str
Arguments: list
EmitValidation: list
Desc: list
@@ -68,15 +64,11 @@ class OpDefinition:
self.OpClass = None
self.OpSize = 0
self.HasSideEffects = False
self.ImplicitFlagClobber = False
self.RAOverride = -1
self.SwitchGen = True
self.ArgPrinter = True
self.SSAArgNum = 0
self.NonSSAArgNum = 0
self.DynamicDispatch = False
self.JITDispatch = True
self.JITDispatchOverride = None
self.Arguments = []
self.EmitValidation = []
self.Desc = []
@@ -152,7 +144,7 @@ def parse_ops(ops):
Argument = Argument.strip()
OpArg = OpArgument()
Split = Argument.split(":", 1)
Split = Argument.split(":")
if len(Split) != 2:
ExitError("Error parsing argument. Missing Type and name colon split")
@@ -221,9 +213,6 @@ def parse_ops(ops):
if "HasSideEffects" in op_val:
OpDef.HasSideEffects = bool(op_val["HasSideEffects"])
if "ImplicitFlagClobber" in op_val:
OpDef.ImplicitFlagClobber = bool(op_val["ImplicitFlagClobber"])
if "ArgPrinter" in op_val:
OpDef.ArgPrinter = bool(op_val["ArgPrinter"])
@@ -239,15 +228,6 @@ def parse_ops(ops):
if "Desc" in op_val:
OpDef.Desc = op_val["Desc"]
if "DynamicDispatch" in op_val:
OpDef.DynamicDispatch = bool(op_val["DynamicDispatch"])
if "JITDispatch" in op_val:
OpDef.JITDispatch = bool(op_val["JITDispatch"])
if "JITDispatchOverride" in op_val:
OpDef.JITDispatchOverride = op_val["JITDispatchOverride"]
# Do some fixups of the data here
if len(OpDef.EmitValidation) != 0:
for i in range(len(OpDef.EmitValidation)):
@@ -377,7 +357,6 @@ def print_ir_sizes():
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool ImplicitFlagClobber(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
output_file.write("#undef IROP_SIZES\n")
@@ -471,17 +450,15 @@ def print_ir_getraargs():
def print_ir_hassideeffects():
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
for array, prop in [("SideEffects", "HasSideEffects"),
("ImplicitFlagClobbers", "ImplicitFlagClobber")]:
output_file.write(f"constexpr std::array<uint8_t, OP_LAST + 1> {array} = {{\n")
for op in IROps:
output_file.write("\t{},\n".format(("true" if getattr(op, prop) else "false")))
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> SideEffects = {\n")
for op in IROps:
output_file.write("\t{},\n".format(("true" if op.HasSideEffects else "false")))
output_file.write("};\n\n")
output_file.write("};\n\n")
output_file.write(f"bool {prop}(IROps Op) {{\n")
output_file.write(f" return {array}[Op];\n")
output_file.write("}\n")
output_file.write("bool HasSideEffects(IROps Op) {\n")
output_file.write(" return SideEffects[Op];\n")
output_file.write("}\n")
output_file.write("#undef IROP_HASSIDEEFFECTS_IMPL\n")
output_file.write("#endif\n\n")
@@ -650,10 +627,6 @@ def print_ir_allocator_helpers():
output_file.write(") {\n")
# Save NZCV if needed before clobbering NZCV
if op.ImplicitFlagClobber:
output_file.write("\t\tSaveNZCV(IROps::OP_{});".format(op.Name.upper()))
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
if op.SSAArgNum != 0:
@@ -702,8 +675,7 @@ def print_ir_allocator_helpers():
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
for Validation in op.EmitValidation:
Sanitized = Validation.replace("\"", "\\\"")
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"\");\n".format(Validation))
output_file.write("\t\t#endif\n")
output_file.write("\t\treturn Op;\n")
@@ -758,38 +730,10 @@ def print_ir_parser_switch_helper():
output_file.write("#undef IROP_PARSER_SWITCH_HELPERS\n")
output_file.write("#endif\n")
def print_ir_dispatcher_defs():
output_dispatch_file.write("#ifdef IROP_DISPATCH_DEFS\n")
for op in IROps:
if op.Name != "Last" and op.SwitchGen and op.JITDispatch and op.JITDispatchOverride == None:
output_dispatch_file.write("DEF_OP({});\n".format(op.Name))
output_dispatch_file.write("#undef IROP_DISPATCH_DEFS\n")
output_dispatch_file.write("#endif\n")
def print_ir_dispatcher_dispatch():
output_dispatch_file.write("#ifdef IROP_DISPATCH_DISPATCH\n")
for op in IROps:
if op.Name != "Last" and op.JITDispatch:
DispatchName = op.Name
if op.JITDispatchOverride != None:
DispatchName = op.JITDispatchOverride
if (op.DynamicDispatch):
output_dispatch_file.write("REGISTER_OP_RT({}, {});\n".format(op.Name.upper(), DispatchName))
else:
output_dispatch_file.write("REGISTER_OP({}, {});\n".format(op.Name.upper(), DispatchName))
output_dispatch_file.write("#undef IROP_DISPATCH_DISPATCH\n")
output_dispatch_file.write("#endif\n")
if (len(sys.argv) < 4):
if (len(sys.argv) < 3):
ExitError()
output_filename = sys.argv[2]
output_dispatcher_filename = sys.argv[3]
json_file = open(sys.argv[1], "r")
json_text = json_file.read()
json_file.close()
@@ -819,10 +763,3 @@ print_ir_allocator_helpers()
print_ir_parser_switch_helper()
output_file.close()
output_dispatch_file = open(output_dispatcher_filename, "w")
print_ir_dispatcher_defs()
print_ir_dispatcher_dispatch()
output_dispatch_file.close()
+61 -27
View File
@@ -7,7 +7,6 @@ set (FEXCORE_BASE_SRCS
Utils/FileLoading.cpp
Utils/ForcedAssert.cpp
Utils/LogManager.cpp
Utils/SpinWaitLock.cpp
)
if (NOT MINGW_BUILD)
@@ -91,6 +90,7 @@ set (SRCS
Interface/Core/CPUBackend.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/GdbServer.cpp
Interface/Core/HostFeatures.cpp
Interface/Core/ObjectCache/JobHandling.cpp
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
@@ -101,23 +101,15 @@ set (SRCS
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher/X87F64.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/SignalDelegator.cpp
Interface/Core/X86Tables.cpp
Interface/Core/X86DebugInfo.cpp
Interface/Core/X86HelperGen.cpp
Interface/Core/ArchHelpers/Arm64Emitter.cpp
Interface/Core/Dispatcher/Dispatcher.cpp
Interface/Core/Dispatcher/X86Dispatcher.cpp
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
Interface/Core/JIT/Arm64/JIT.cpp
Interface/Core/JIT/Arm64/ALUOps.cpp
Interface/Core/JIT/Arm64/AtomicOps.cpp
Interface/Core/JIT/Arm64/BranchOps.cpp
Interface/Core/JIT/Arm64/ConversionOps.cpp
Interface/Core/JIT/Arm64/EncryptionOps.cpp
Interface/Core/JIT/Arm64/FlagOps.cpp
Interface/Core/JIT/Arm64/MemoryOps.cpp
Interface/Core/JIT/Arm64/MiscOps.cpp
Interface/Core/JIT/Arm64/MoveOps.cpp
Interface/Core/JIT/Arm64/VectorOps.cpp
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
Interface/Core/X86Tables/BaseTables.cpp
Interface/Core/X86Tables/DDDTables.cpp
Interface/Core/X86Tables/EVEXTables.cpp
@@ -149,7 +141,8 @@ set (SRCS
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
Interface/IR/Passes/DeadStoreElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/InlineCallOptimization.cpp
Interface/IR/Passes/SyscallOptimization.cpp
Utils/NetStream.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/Profiler.cpp
@@ -166,7 +159,24 @@ if (ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
Utils/AllocatorOverride.cpp)
endif()
set(DEFINES -DTHREAD_LOCAL=_Thread_local -DJIT_ARM64)
if (ENABLE_INTERPRETER)
list(APPEND SRCS
Interface/Core/Interpreter/InterpreterCore.cpp
Interface/Core/Interpreter/InterpreterOps.cpp
Interface/Core/Interpreter/ALUOps.cpp
Interface/Core/Interpreter/AtomicOps.cpp
Interface/Core/Interpreter/BranchOps.cpp
Interface/Core/Interpreter/ConversionOps.cpp
Interface/Core/Interpreter/EncryptionOps.cpp
Interface/Core/Interpreter/F80Ops.cpp
Interface/Core/Interpreter/FlagOps.cpp
Interface/Core/Interpreter/MemoryOps.cpp
Interface/Core/Interpreter/MiscOps.cpp
Interface/Core/Interpreter/MoveOps.cpp
Interface/Core/Interpreter/VectorOps.cpp)
endif()
set(DEFINES -DTHREAD_LOCAL=_Thread_local)
if (_M_X86_64)
list(APPEND DEFINES -D_M_X86_64=1)
@@ -185,16 +195,43 @@ if (ENABLE_VIXL_DISASSEMBLER)
list(APPEND DEFINES -DVIXL_DISASSEMBLER=1)
endif()
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=__attribute__((preserve_all));-DFEXCORE_HAS_PRESERVE_ALL_ATTR=1")
else()
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=;-DFEXCORE_HAS_PRESERVE_ALL_ATTR=0")
if (ENABLE_JIT_X86_64)
list(APPEND SRCS
Interface/Core/JIT/x86_64/JIT.cpp
Interface/Core/JIT/x86_64/ALUOps.cpp
Interface/Core/JIT/x86_64/AtomicOps.cpp
Interface/Core/JIT/x86_64/BranchOps.cpp
Interface/Core/JIT/x86_64/ConversionOps.cpp
Interface/Core/JIT/x86_64/EncryptionOps.cpp
Interface/Core/JIT/x86_64/FlagOps.cpp
Interface/Core/JIT/x86_64/MemoryOps.cpp
Interface/Core/JIT/x86_64/MiscOps.cpp
Interface/Core/JIT/x86_64/MoveOps.cpp
Interface/Core/JIT/x86_64/VectorOps.cpp
Interface/Core/JIT/x86_64/x64Relocations.cpp
)
list(APPEND DEFINES -DJIT_X86_64)
endif()
# Some defines for the softfloat library
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
if (ENABLE_JIT_ARM64)
list(APPEND DEFINES -DJIT_ARM64)
list(APPEND SRCS
Interface/Core/JIT/Arm64/JIT.cpp
Interface/Core/JIT/Arm64/ALUOps.cpp
Interface/Core/JIT/Arm64/AtomicOps.cpp
Interface/Core/JIT/Arm64/BranchOps.cpp
Interface/Core/JIT/Arm64/ConversionOps.cpp
Interface/Core/JIT/Arm64/EncryptionOps.cpp
Interface/Core/JIT/Arm64/FlagOps.cpp
Interface/Core/JIT/Arm64/MemoryOps.cpp
Interface/Core/JIT/Arm64/MiscOps.cpp
Interface/Core/JIT/Arm64/MoveOps.cpp
Interface/Core/JIT/Arm64/VectorOps.cpp
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
)
endif()
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils)
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
if (NOT MINGW_BUILD)
list (APPEND LIBS dl)
@@ -218,16 +255,15 @@ configure_file(
# Generate IR include file
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
set(OUTPUT_DISPATCHER_NAME "${OUTPUT_IR_FOLDER}/IRDefines_Dispatch.inc")
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
add_custom_command(
OUTPUT "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
OUTPUT "${OUTPUT_NAME}"
DEPENDS "${INPUT_NAME}"
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
)
set_source_files_properties(${OUTPUT_NAME} PROPERTIES
@@ -359,7 +395,6 @@ function(AddObject Name Type)
add_library(${Name} ${Type} ${SRCS})
target_link_libraries(${Name} FEXCore_Base)
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
AddDefaultOptionsToTarget(${Name})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
@@ -368,7 +403,6 @@ endfunction()
function(AddLibrary Name Type)
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
target_link_libraries(${Name} FEXCore_Base)
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
if (MINGW_BUILD)
# Mingw build isn't building a linux shared library, so it can't have a SONAME.
-1
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/Allocator.h>
+36 -86
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/fextl/fmt.h>
#include "Common/JitSymbols.h"
@@ -27,6 +26,42 @@ namespace FEXCore {
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
}
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
@@ -51,89 +86,4 @@ namespace FEXCore {
}
}
// Buffered JIT symbols.
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, GuestAddr, CodeSize);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, CodeSize, Name, Offset);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}\n", HostAddr, CodeSize, Name);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
RegisterNamedRegion(Buffer, HostAddr, CodeSize, Name);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite) {
auto Now = std::chrono::steady_clock::now();
if (!ForceWrite) {
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) &&
Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
// Still buffering, no need to write.
return;
}
}
Buffer->LastWrite = Now;
auto Result = write(fd, Buffer->Buffer, Buffer->Offset);
if (Result == -1 && errno == EBADF) {
fd = -1;
}
Buffer->Offset = 0;
}
} // namespace FEXCore
+3 -15
View File
@@ -1,10 +1,5 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/fextl/memory.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <chrono>
#include <cstdint>
#include <cstdio>
#include <memory>
@@ -17,20 +12,13 @@ public:
~JITSymbols();
void InitFile();
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
// Allocate JIT buffer.
static fextl::unique_ptr<Core::JITSymbolBuffer> AllocateBuffer() {
return fextl::make_unique<Core::JITSymbolBuffer>();
}
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name);
private:
int fd{-1};
void WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite = false);
};
}
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_add( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_div( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
bool extF80_eq( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
bool extF80_lt( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_mul( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_rem( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
extF80_roundToInt( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_sqrt( extFloat80_t a )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_sub( extFloat80_t a, extFloat80_t b )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float128_t extF80_to_f128( extFloat80_t a )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float32_t extF80_to_f32( extFloat80_t a )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float64_t extF80_to_f64( extFloat80_t a )
{
union { struct extFloat80M s; extFloat80_t f; } uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
int_fast32_t
extF80_to_i32( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
int_fast64_t
extF80_to_i64( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
uint_fast64_t
extF80_to_ui64( extFloat80_t a, uint_fast8_t roundingMode, bool exact )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f128_to_extF80( float128_t a )
{
union ui128_f128 uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f32_to_extF80( float32_t a )
{
union ui32_f32 uA;
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f64_to_extF80( float64_t a )
{
union ui64_f64 uA;
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t i32_to_extF80( int32_t a )
{
uint_fast16_t uiZ64;
@@ -68,11 +68,9 @@ uint_fast64_t
uint_fast64_t softfloat_roundMToUI64( bool, uint32_t *, uint_fast8_t, bool );
#endif
FEXCORE_PRESERVE_ALL_ATTR
int_fast32_t softfloat_roundToI32( bool, uint_fast64_t, uint_fast8_t, bool );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
int_fast64_t
softfloat_roundToI64(
bool, uint_fast64_t, uint_fast64_t, uint_fast8_t, bool );
@@ -111,10 +109,8 @@ float16_t
#define isNaNF32UI( a ) (((~(a) & 0x7F800000) == 0) && ((a) & 0x007FFFFF))
struct exp16_sig32 { int_fast16_t exp; uint_fast32_t sig; };
FEXCORE_PRESERVE_ALL_ATTR
struct exp16_sig32 softfloat_normSubnormalF32Sig( uint_fast32_t );
FEXCORE_PRESERVE_ALL_ATTR
float32_t softfloat_roundPackToF32( bool, int_fast16_t, uint_fast32_t );
float32_t softfloat_normRoundPackToF32( bool, int_fast16_t, uint_fast32_t );
@@ -134,10 +130,8 @@ float32_t
#define isNaNF64UI( a ) (((~(a) & UINT64_C( 0x7FF0000000000000 )) == 0) && ((a) & UINT64_C( 0x000FFFFFFFFFFFFF )))
struct exp16_sig64 { int_fast16_t exp; uint_fast64_t sig; };
FEXCORE_PRESERVE_ALL_ATTR
struct exp16_sig64 softfloat_normSubnormalF64Sig( uint_fast64_t );
FEXCORE_PRESERVE_ALL_ATTR
float64_t softfloat_roundPackToF64( bool, int_fast16_t, uint_fast64_t );
float64_t softfloat_normRoundPackToF64( bool, int_fast16_t, uint_fast64_t );
@@ -161,14 +155,11 @@ float64_t
*----------------------------------------------------------------------------*/
struct exp32_sig64 { int_fast32_t exp; uint64_t sig; };
FEXCORE_PRESERVE_ALL_ATTR
struct exp32_sig64 softfloat_normSubnormalExtF80Sig( uint_fast64_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
softfloat_roundPackToExtF80(
bool, int_fast32_t, uint_fast64_t, uint_fast64_t, uint_fast8_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
softfloat_normRoundPackToExtF80(
bool, int_fast32_t, uint_fast64_t, uint_fast64_t, uint_fast8_t );
@@ -190,7 +181,6 @@ extFloat80_t
#define isNaNF128UI( a64, a0 ) (((~(a64) & UINT64_C( 0x7FFF000000000000 )) == 0) && (a0 || ((a64) & UINT64_C( 0x0000FFFFFFFFFFFF ))))
struct exp32_sig128 { int_fast32_t exp; struct uint128 sig; };
FEXCORE_PRESERVE_ALL_ATTR
struct exp32_sig128
softfloat_normSubnormalF128Sig( uint_fast64_t, uint_fast64_t );
@@ -53,7 +53,6 @@ INLINE
uint64_t softfloat_shortShiftRightJam64( uint64_t a, uint_fast8_t dist )
{ return a>>dist | ((a & (((uint_fast64_t) 1<<dist) - 1)) != 0); }
#else
FEXCORE_PRESERVE_ALL_ATTR
uint64_t softfloat_shortShiftRightJam64( uint64_t a, uint_fast8_t dist );
#endif
#endif
@@ -75,7 +74,6 @@ INLINE uint32_t softfloat_shiftRightJam32( uint32_t a, uint_fast16_t dist )
(dist < 31) ? a>>dist | ((uint32_t) (a<<(-dist & 31)) != 0) : (a != 0);
}
#else
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_shiftRightJam32( uint32_t a, uint_fast16_t dist );
#endif
#endif
@@ -97,7 +95,6 @@ INLINE uint64_t softfloat_shiftRightJam64( uint64_t a, uint_fast32_t dist )
(dist < 63) ? a>>dist | ((uint64_t) (a<<(-dist & 63)) != 0) : (a != 0);
}
#else
FEXCORE_PRESERVE_ALL_ATTR
uint64_t softfloat_shiftRightJam64( uint64_t a, uint_fast32_t dist );
#endif
#endif
@@ -151,7 +148,6 @@ INLINE uint_fast8_t softfloat_countLeadingZeros32( uint32_t a )
return count;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
uint_fast8_t softfloat_countLeadingZeros32( uint32_t a );
#endif
#endif
@@ -161,7 +157,6 @@ uint_fast8_t softfloat_countLeadingZeros32( uint32_t a );
| Returns the number of leading 0 bits before the most-significant 1 bit of
| 'a'. If 'a' is zero, 64 is returned.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast8_t softfloat_countLeadingZeros64( uint64_t a );
#endif
@@ -183,7 +178,6 @@ extern const uint16_t softfloat_approxRecip_1k1s[16];
#ifdef SOFTFLOAT_FAST_DIV64TO32
#define softfloat_approxRecip32_1( a ) ((uint32_t) (UINT64_C( 0x7FFFFFFFFFFFFFFF ) / (uint32_t) (a)))
#else
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_approxRecip32_1( uint32_t a );
#endif
#endif
@@ -210,7 +204,6 @@ extern const uint16_t softfloat_approxRecipSqrt_1k1s[16];
| returned is also always within the range 0.5 to 1; thus, the most-
| significant bit of the result is always set.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_approxRecipSqrt32_1( unsigned int oddExpA, uint32_t a );
#endif
@@ -247,7 +240,6 @@ INLINE
bool softfloat_le128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{ return (a64 < b64) || ((a64 == b64) && (a0 <= b0)); }
#else
FEXCORE_PRESERVE_ALL_ATTR
bool softfloat_le128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 );
#endif
#endif
@@ -263,7 +255,6 @@ INLINE
bool softfloat_lt128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{ return (a64 < b64) || ((a64 == b64) && (a0 < b0)); }
#else
FEXCORE_PRESERVE_ALL_ATTR
bool softfloat_lt128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 );
#endif
#endif
@@ -284,7 +275,6 @@ struct uint128
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_shortShiftLeft128( uint64_t a64, uint64_t a0, uint_fast8_t dist );
#endif
@@ -306,7 +296,6 @@ struct uint128
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_shortShiftRight128( uint64_t a64, uint64_t a0, uint_fast8_t dist );
#endif
@@ -424,7 +413,6 @@ struct uint64_extra
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint64_extra
softfloat_shiftRightJam64Extra(
uint64_t a, uint64_t extra, uint_fast32_t dist );
@@ -504,7 +492,6 @@ struct uint128
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_add128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 );
#endif
@@ -541,7 +528,6 @@ struct uint128
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_sub128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 );
#endif
@@ -576,7 +562,6 @@ INLINE struct uint128 softfloat_mul64ByShifted32To128( uint64_t a, uint32_t b )
return z;
}
#else
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_mul64ByShifted32To128( uint64_t a, uint32_t b );
#endif
#endif
@@ -585,7 +570,6 @@ struct uint128 softfloat_mul64ByShifted32To128( uint64_t a, uint32_t b );
/*----------------------------------------------------------------------------
| Returns the 128-bit product of 'a' and 'b'.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_mul64To128( uint64_t a, uint64_t b );
#endif
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_add128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_add128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
extern const uint16_t softfloat_approxRecip_1k0s[16];
extern const uint16_t softfloat_approxRecip_1k1s[16];
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_approxRecip32_1( uint32_t a )
{
int index;
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
extern const uint16_t softfloat_approxRecipSqrt_1k0s[];
extern const uint16_t softfloat_approxRecipSqrt_1k1s[];
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_approxRecipSqrt32_1( unsigned int oddExpA, uint32_t a )
{
int index;
@@ -44,7 +44,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| floating-point NaN, and returns the bit pattern of this value as an unsigned
| integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_commonNaNToExtF80UI( const struct commonNaN *aPtr )
{
struct uint128 uiZ;
@@ -43,7 +43,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| Converts the common NaN pointed to by `aPtr' into a 128-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_commonNaNToF128UI( const struct commonNaN *aPtr )
{
struct uint128 uiZ;
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| Converts the common NaN pointed to by `aPtr' into a 32-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast32_t softfloat_commonNaNToF32UI( const struct commonNaN *aPtr )
{
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| Converts the common NaN pointed to by `aPtr' into a 64-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast64_t softfloat_commonNaNToF64UI( const struct commonNaN *aPtr )
{
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#define softfloat_countLeadingZeros32 softfloat_countLeadingZeros32
#include "primitives.h"
FEXCORE_PRESERVE_ALL_ATTR
uint_fast8_t softfloat_countLeadingZeros32( uint32_t a )
{
uint_fast8_t count;
@@ -42,7 +42,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#define softfloat_countLeadingZeros64 softfloat_countLeadingZeros64
#include "primitives.h"
FEXCORE_PRESERVE_ALL_ATTR
uint_fast8_t softfloat_countLeadingZeros64( uint64_t a )
{
uint_fast8_t count;
@@ -46,7 +46,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| location pointed to by `zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void
softfloat_extF80UIToCommonNaN(
uint_fast16_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr )
@@ -47,7 +47,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| pointed to by `zPtr'. If the NaN is a signaling NaN, the invalid exception
| is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void
softfloat_f128UIToCommonNaN(
uint_fast64_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr )
@@ -45,7 +45,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| location pointed to by `zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_f32UIToCommonNaN( uint_fast32_t uiA, struct commonNaN *zPtr )
{
@@ -45,7 +45,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| location pointed to by `zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_f64UIToCommonNaN( uint_fast64_t uiA, struct commonNaN *zPtr )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_le128
FEXCORE_PRESERVE_ALL_ATTR
bool softfloat_le128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_lt128
FEXCORE_PRESERVE_ALL_ATTR
bool softfloat_lt128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_mul64ByShifted32To128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_mul64ByShifted32To128( uint64_t a, uint32_t b )
{
uint_fast64_t mid;
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_mul64To128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_mul64To128( uint64_t a, uint64_t b )
{
uint32_t a32, a0, b32, b0;
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
softfloat_normRoundPackToExtF80(
bool sign,
@@ -38,7 +38,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
struct exp32_sig64 softfloat_normSubnormalExtF80Sig( uint_fast64_t sig )
{
int_fast8_t shiftDist;
@@ -38,7 +38,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
struct exp32_sig128
softfloat_normSubnormalF128Sig( uint_fast64_t sig64, uint_fast64_t sig0 )
{
@@ -38,7 +38,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
struct exp16_sig32 softfloat_normSubnormalF32Sig( uint_fast32_t sig )
{
int_fast8_t shiftDist;
@@ -38,7 +38,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "platform.h"
#include "internals.h"
FEXCORE_PRESERVE_ALL_ATTR
struct exp16_sig64 softfloat_normSubnormalF64Sig( uint_fast64_t sig )
{
int_fast8_t shiftDist;
@@ -50,7 +50,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| result. If either original floating-point value is a signaling NaN, the
| invalid exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_propagateNaNExtF80UI(
uint_fast16_t uiA64,
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t
softfloat_roundPackToExtF80(
bool sign,
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float32_t
softfloat_roundPackToF32( bool sign, int_fast16_t exp, uint_fast32_t sig )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
float64_t
softfloat_roundPackToF64( bool sign, int_fast16_t exp, uint_fast64_t sig )
{
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
int_fast32_t
softfloat_roundToI32(
bool sign, uint_fast64_t sig, uint_fast8_t roundingMode, bool exact )
@@ -41,7 +41,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "specialize.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
int_fast64_t
softfloat_roundToI64(
bool sign,
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shiftRightJam32
FEXCORE_PRESERVE_ALL_ATTR
uint32_t softfloat_shiftRightJam32( uint32_t a, uint_fast16_t dist )
{
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shiftRightJam64
FEXCORE_PRESERVE_ALL_ATTR
uint64_t softfloat_shiftRightJam64( uint64_t a, uint_fast32_t dist )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shiftRightJam64Extra
FEXCORE_PRESERVE_ALL_ATTR
struct uint64_extra
softfloat_shiftRightJam64Extra(
uint64_t a, uint64_t extra, uint_fast32_t dist )
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shortShiftLeft128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_shortShiftLeft128( uint64_t a64, uint64_t a0, uint_fast8_t dist )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shortShiftRight128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_shortShiftRight128( uint64_t a64, uint64_t a0, uint_fast8_t dist )
{
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_shortShiftRightJam64
FEXCORE_PRESERVE_ALL_ATTR
uint64_t softfloat_shortShiftRightJam64( uint64_t a, uint_fast8_t dist )
{
@@ -40,7 +40,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#ifndef softfloat_sub128
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_sub128( uint64_t a64, uint64_t a0, uint64_t b64, uint64_t b0 )
{
@@ -92,7 +92,6 @@ enum {
/*----------------------------------------------------------------------------
| Routine to raise any or all of the software floating-point exception flags.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_raiseFlags( uint_fast8_t );
/*----------------------------------------------------------------------------
@@ -111,7 +110,6 @@ float16_t ui64_to_f16( uint64_t );
float32_t ui64_to_f32( uint64_t );
float64_t ui64_to_f64( uint64_t );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t ui64_to_extF80( uint64_t );
float128_t ui64_to_f128( uint64_t );
#endif
@@ -121,7 +119,6 @@ float16_t i32_to_f16( int32_t );
float32_t i32_to_f32( int32_t );
float64_t i32_to_f64( int32_t );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t i32_to_extF80( int32_t );
float128_t i32_to_f128( int32_t );
#endif
@@ -186,7 +183,6 @@ int_fast64_t f32_to_i64_r_minMag( float32_t, bool );
float16_t f32_to_f16( float32_t );
float64_t f32_to_f64( float32_t );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f32_to_extF80( float32_t );
float128_t f32_to_f128( float32_t );
#endif
@@ -222,7 +218,6 @@ int_fast64_t f64_to_i64_r_minMag( float64_t, bool );
float16_t f64_to_f16( float64_t );
float32_t f64_to_f32( float64_t );
#ifdef SOFTFLOAT_FAST_INT64
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f64_to_extF80( float64_t );
float128_t f64_to_f128( float64_t );
#endif
@@ -255,41 +250,26 @@ extern THREAD_LOCAL uint_fast8_t extF80_roundingPrecision;
*----------------------------------------------------------------------------*/
#ifdef SOFTFLOAT_FAST_INT64
uint_fast32_t extF80_to_ui32( extFloat80_t, uint_fast8_t, bool );
FEXCORE_PRESERVE_ALL_ATTR
uint_fast64_t extF80_to_ui64( extFloat80_t, uint_fast8_t, bool );
FEXCORE_PRESERVE_ALL_ATTR
int_fast32_t extF80_to_i32( extFloat80_t, uint_fast8_t, bool );
FEXCORE_PRESERVE_ALL_ATTR
int_fast64_t extF80_to_i64( extFloat80_t, uint_fast8_t, bool );
uint_fast32_t extF80_to_ui32_r_minMag( extFloat80_t, bool );
uint_fast64_t extF80_to_ui64_r_minMag( extFloat80_t, bool );
int_fast32_t extF80_to_i32_r_minMag( extFloat80_t, bool );
int_fast64_t extF80_to_i64_r_minMag( extFloat80_t, bool );
float16_t extF80_to_f16( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
float32_t extF80_to_f32( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
float64_t extF80_to_f64( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
float128_t extF80_to_f128( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_roundToInt( extFloat80_t, uint_fast8_t, bool );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_add( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_sub( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_mul( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_div( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_rem( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t extF80_sqrt( extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
bool extF80_eq( extFloat80_t, extFloat80_t );
bool extF80_le( extFloat80_t, extFloat80_t );
FEXCORE_PRESERVE_ALL_ATTR
bool extF80_lt( extFloat80_t, extFloat80_t );
bool extF80_eq_signaling( extFloat80_t, extFloat80_t );
bool extF80_le_quiet( extFloat80_t, extFloat80_t );
@@ -340,7 +320,6 @@ int_fast64_t f128_to_i64_r_minMag( float128_t, bool );
float16_t f128_to_f16( float128_t );
float32_t f128_to_f32( float128_t );
float64_t f128_to_f64( float128_t );
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t f128_to_extF80( float128_t );
float128_t f128_roundToInt( float128_t, uint_fast8_t, bool );
float128_t f128_add( float128_t, float128_t );
@@ -43,7 +43,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
| to substitute a result value. If traps are not implemented, this routine
| should be simply `softfloat_exceptionFlags |= flags;'.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_raiseFlags( uint_fast8_t flags )
{
@@ -135,14 +135,12 @@ uint_fast16_t
| location pointed to by 'zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_f32UIToCommonNaN( uint_fast32_t uiA, struct commonNaN *zPtr );
/*----------------------------------------------------------------------------
| Converts the common NaN pointed to by 'aPtr' into a 32-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast32_t softfloat_commonNaNToF32UI( const struct commonNaN *aPtr );
/*----------------------------------------------------------------------------
@@ -172,14 +170,12 @@ uint_fast32_t
| location pointed to by 'zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void softfloat_f64UIToCommonNaN( uint_fast64_t uiA, struct commonNaN *zPtr );
/*----------------------------------------------------------------------------
| Converts the common NaN pointed to by 'aPtr' into a 64-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
uint_fast64_t softfloat_commonNaNToF64UI( const struct commonNaN *aPtr );
/*----------------------------------------------------------------------------
@@ -219,7 +215,6 @@ uint_fast64_t
| location pointed to by 'zPtr'. If the NaN is a signaling NaN, the invalid
| exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void
softfloat_extF80UIToCommonNaN(
uint_fast16_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr );
@@ -229,7 +224,6 @@ void
| floating-point NaN, and returns the bit pattern of this value as an unsigned
| integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_commonNaNToExtF80UI( const struct commonNaN *aPtr );
/*----------------------------------------------------------------------------
@@ -241,7 +235,6 @@ struct uint128 softfloat_commonNaNToExtF80UI( const struct commonNaN *aPtr );
| result. If either original floating-point value is a signaling NaN, the
| invalid exception is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128
softfloat_propagateNaNExtF80UI(
uint_fast16_t uiA64,
@@ -271,7 +264,6 @@ struct uint128
| pointed to by 'zPtr'. If the NaN is a signaling NaN, the invalid exception
| is raised.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
void
softfloat_f128UIToCommonNaN(
uint_fast64_t uiA64, uint_fast64_t uiA0, struct commonNaN *zPtr );
@@ -280,7 +272,6 @@ void
| Converts the common NaN pointed to by 'aPtr' into a 128-bit floating-point
| NaN, and returns the bit pattern of this value as an unsigned integer.
*----------------------------------------------------------------------------*/
FEXCORE_PRESERVE_ALL_ATTR
struct uint128 softfloat_commonNaNToF128UI( const struct commonNaN * );
/*----------------------------------------------------------------------------
@@ -39,7 +39,6 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
#include "internals.h"
#include "softfloat.h"
FEXCORE_PRESERVE_ALL_ATTR
extFloat80_t ui64_to_extF80( uint64_t a )
{
uint_fast16_t uiZ64;
+1 -21
View File
@@ -1,10 +1,9 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <cmath>
#include <cstring>
@@ -63,7 +62,6 @@ struct FEX_PACKED X80SoftFloat {
}
// Ops
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
@@ -85,7 +83,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
@@ -107,7 +104,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
@@ -129,7 +125,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
@@ -151,7 +146,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
@@ -174,7 +168,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
@@ -197,17 +190,14 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
return extF80_roundToInt(lhs, RoundMode, false);
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
@@ -231,7 +221,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
@@ -253,14 +242,12 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
*eq = extF80_eq(lhs, rhs);
*lt = extF80_lt(lhs, rhs);
*nan = IsNan(lhs) || IsNan(rhs);
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -289,7 +276,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -313,7 +299,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -339,7 +324,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -365,7 +349,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -389,7 +372,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -412,7 +394,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
@@ -435,7 +416,6 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
-1
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/fextl/string.h>
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/fextl/string.h>
+24 -5
View File
@@ -1,5 +1,5 @@
// SPDX-License-Identifier: MIT
#include "Common/StringConv.h"
#include "Common/StringUtils.h"
#include "FEXCore/Utils/EnumUtils.h"
#include <FEXCore/Config/Config.h>
@@ -7,7 +7,6 @@
#include <FEXCore/Utils/CPUInfo.h>
#include <FEXCore/Utils/FileLoading.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/StringUtils.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/list.h>
#include <FEXCore/fextl/map.h>
@@ -321,15 +320,30 @@ namespace DefaultValues {
Meta->Load();
// Do configuration option fix ups after everything is reloaded
{
// Always fix up the number of threads and create the configuration
// Otherwise the application could receive zero as the number of threads
FEX_CONFIG_OPT(Cores, THREADS);
if (Cores == 0) {
// When the number of emulated CPU cores is zero then auto detect
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, fextl::fmt::format("{}", FEXCore::CPUInfo::CalculateNumberOfCPUs()));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
// Sanitize Core option
FEX_CONFIG_OPT(Core, CORE);
#if (_M_X86_64)
constexpr uint32_t MaxCoreNumber = 1;
constexpr uint32_t MaxCoreNumber = 2;
#else
constexpr uint32_t MaxCoreNumber = 0;
constexpr uint32_t MaxCoreNumber = 1;
#endif
if (Core > MaxCoreNumber) {
#ifdef INTERPRETER_ENABLED
constexpr uint32_t MinCoreNumber = 0;
#else
constexpr uint32_t MinCoreNumber = 1;
#endif
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
// Sanitize the core option by setting the core to the JIT if invalid
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
}
@@ -338,6 +352,11 @@ namespace DefaultValues {
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(Core, CORE);
if (CacheObjectCodeCompilation() && Core() == FEXCore::Config::CONFIG_INTERPRETER) {
// If running the interpreter then disable cache code compilation
FEXCore::Config::Erase(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION);
}
}
fextl::string ContainerPrefix { FindContainerPrefix() };
+30 -51
View File
@@ -6,12 +6,12 @@
"Default": "FEXCore::Config::ConfigCore::CONFIG_IRJIT",
"TextDefault": "irjit",
"ShortArg": "c",
"Choices": [ "irjit", "host" ],
"Choices": [ "irint", "irjit", "host" ],
"ArgumentHandler": "CoreHandler",
"Desc": [
"Which CPU core to use",
"host only exists on x86_64",
"[irjit, host]"
"[irint, irjit, host]"
]
},
"Multiblock": {
@@ -31,6 +31,15 @@
"Maximum number of instruction to store in a block"
]
},
"Threads": {
"Type": "uint32",
"Default": "0",
"ShortArg": "T",
"Desc": [
"Number of physical hardware threads to tell the process we have.",
"0 will auto detect."
]
},
"CacheObjectCodeCompilation": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
@@ -50,8 +59,6 @@
"DISABLESVE": "disablesve",
"ENABLEAVX": "enableavx",
"DISABLEAVX": "disableavx",
"ENABLEAVX2": "enableavx2",
"DISABLEAVX2": "disableavx2",
"ENABLEAFP": "enableafp",
"DISABLEAFP": "disableafp",
"ENABLELRCPC": "enablelrcpc",
@@ -69,24 +76,13 @@
"ENABLEATOMICS": "enableatomics",
"DISABLEATOMICS": "disableatomics",
"ENABLEFCMA": "enablefcma",
"DISABLEFCMA": "disablefcma",
"ENABLEFLAGM": "enableflagm",
"DISABLEFLAGM": "disableflagm",
"ENABLEFLAGM2": "enableflagm2",
"DISABLEFLAGM2": "disableflagm2",
"ENABLECRYPTO": "enablecrypto",
"DISABLECRYPTO": "disablecrypto",
"ENABLERPRES": "enablerpres",
"DISABLERPRES": "disablerpres",
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
"DISABLEPRESERVEALLABI": "disablepreserveallabi"
"DISABLEFCMA": "disablefcma"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
"\toff: Default CPU features queried from CPU features",
"\t{enable,disable}sve: Will force enable or disable sve even if the host doesn't support it",
"\t{enable,disable}avx: Will force enable or disable avx even if the host doesn't support it",
"\t{enable,disable}avx2: Will force enable or disable avx2 even if the host doesn't support it",
"\t{enable,disable}afp: Will force enable or disable afp even if the host doesn't support it",
"\t{enable,disable}lrcpc: Will force enable or disable lrcpc even if the host doesn't support it",
"\t{enable,disable}lrcpc2: Will force enable or disable lrcpc2 even if the host doesn't support it",
@@ -95,32 +91,7 @@
"\t{enable,disable}rng: Will force enable or disable rng even if the host doesn't support it",
"\t{enable,disable}clzero: Will force enable or disable clzero even if the host doesn't support it",
"\t{enable,disable}atomics: Will force enable or disable ARMv8.1 LSE atomics even if the host doesn't support it",
"\t{enable,disable}fcma: Will force enable or disable fcma even if the host doesn't support it",
"\t{enable,disable}flagm: Will force enable or disable flagm even if the host doesn't support it",
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it",
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
]
},
"CPUID": {
"Type": "strenum",
"Default": "FEXCore::Config::CPUID::OFF",
"Enums": {
"ENABLESHA": "enablesha",
"DISABLESHA": "disablesha"
},
"Desc": [
"Allows controlling of the CPU features are exposed in CPUID.",
"\toff: Default CPU features queried from CPU features",
"\t{enable,disable}sha: Will force enable or disable sha even if the host doesn't support it"
]
},
"SmallTSCScale": {
"Type": "bool",
"Default": "true",
"Desc": [
"Scales the cycle counter on systems that have low frequencies."
"\t{enable,disable}fcma: Will force enable or disable fcma even if the host doesn't support it"
]
}
},
@@ -270,6 +241,23 @@
"Disables optimizations passes for debugging."
]
},
"SRA": {
"Type": "bool",
"Default": "true",
"Desc": [
"Set to false to disable Static Register Allocation"
]
},
"Force32BitAllocator": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces use of the 32-bit allocator on 32-bit applications",
"Used to work around ulimit problems of CI runner",
"Potentially useful for debugging memory problems",
"32-bit allocator is always used if your host kernel is older than 4.17"
]
},
"GlobalJITNaming": {
"Type": "bool",
"Default": "false",
@@ -502,15 +490,6 @@
"IS64BIT_MODE": {
"Type": "bool",
"Default": "false"
},
"DISABLE_VIXL_INDIRECT_RUNTIME_CALLS": {
"Type": "bool",
"Default": "true",
"Desc": [
"This option is used for the InstructionCountCI so it can generate the same codegen between Arm64 hosts and vixl simulator hosts.",
"Vixl simulator indirect runtime calls are a special hlt instruction with metadata after it. Effectively making a custom call instruction.",
"With visual simulator calls disabled, the code generation would be the same as on a native Arm64 host, but running the code is broken."
]
}
}
}
+28 -3
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#include "Interface/Context/Context.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
@@ -12,6 +11,10 @@
#include <string.h>
#include <utility>
namespace FEXCore::HLE {
class SyscallVisitor;
}
namespace FEXCore::Context {
void InitializeStaticTables(OperatingMode Mode) {
X86Tables::InitializeInfoTables(Mode);
@@ -22,6 +25,12 @@ namespace FEXCore::Context {
return fextl::make_unique<FEXCore::Context::ContextImpl>();
}
bool FEXCore::Context::ContextImpl::InitializeContext() {
// This should be used for generating things that are shared between threads
CPUID.Init(this);
return true;
}
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
CustomExitHandler = std::move(handler);
}
@@ -30,12 +39,28 @@ namespace FEXCore::Context {
return CustomExitHandler;
}
void FEXCore::Context::ContextImpl::Stop() {
Stop(false);
}
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
CompileBlock(Thread->CurrentFrame, GuestRIP);
}
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
FEXCore::Context::ExitReason FEXCore::Context::ContextImpl::GetExitReason() {
return ParentThread->ExitReason;
}
bool FEXCore::Context::ContextImpl::IsDone() const {
return IsPaused();
}
void FEXCore::Context::ContextImpl::GetCPUState(FEXCore::Core::CPUState *State) const {
memcpy(State, ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
}
void FEXCore::Context::ContextImpl::SetCPUState(const FEXCore::Core::CPUState *State) {
memcpy(ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
}
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
+112 -69
View File
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Common/JitSymbols.h"
@@ -14,8 +13,8 @@
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/DeferredSignalMutex.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
@@ -37,6 +36,7 @@
namespace FEXCore {
class CodeLoader;
class ThunkHandler;
class GdbServer;
namespace CodeSerialize {
class CodeObjectSerializeService;
@@ -44,6 +44,8 @@ namespace CodeSerialize {
namespace CPU {
class Arm64JITCore;
class X86JITCore;
class InterpreterCore;
class Dispatcher;
}
namespace HLE {
@@ -68,29 +70,33 @@ namespace FEXCore::Context {
MODE_SINGLESTEP = 1,
};
struct ExitFunctionLinkData {
uint64_t HostBranch;
uint64_t GuestRIP;
};
using BlockDelinkerFunc = void(*)(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record);
constexpr uint32_t TSC_SCALE = 128;
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
class ContextImpl final : public FEXCore::Context::Context {
public:
// Context base class implementation.
bool InitCore() override;
bool InitializeContext() override;
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
void SetExitHandler(ExitHandler handler) override;
ExitHandler GetExitHandler() const override;
ExitReason RunUntilExit(FEXCore::Core::InternalThreadState *Thread) override;
void Pause() override;
void Run() override;
void Stop() override;
void Step() override;
void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) override;
ExitReason RunUntilExit() override;
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
int GetProgramStatus() const override;
ExitReason GetExitReason() override;
bool IsDone() const override;
void GetCPUState(FEXCore::Core::CPUState *State) const override;
void SetCPUState(const FEXCore::Core::CPUState *State) override;
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
@@ -99,48 +105,55 @@ namespace FEXCore::Context {
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) override;
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) override;
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param InitialRIP The starting RIP of this thread
* @param StackPointer The starting RSP of this thread
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
* @param NewThreadState The initial thread state to setup for our state
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* Parent thread Creation:
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
* - CTX->RunUntilExit(Thread);
* OS thread Creation:
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
* - ThreadHandler calls `CTX->ExecutionThread(Thread)`
* - Thread = CreateThread(NewState, PPID);
* - InitializeThread(Thread);
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
* - Thread = CreateThread(CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread = CreateThread(NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
*
* @param Thread The internal FEX thread state object
*
* The OS thread will wait until RunThread is executed
*/
void InitializeThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Starts the OS thread object to start executing guest code
*
* @param Thread The internal FEX thread state object
*/
void RunThread(FEXCore::Core::InternalThreadState *Thread) override;
void StopThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Destroys this FEX thread object and stops tracking it internally
*
* @param Thread The internal FEX thread state object
*/
void DestroyThread(FEXCore::Core::InternalThreadState *Thread, bool NeedsTLSUninstall) override;
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
#ifndef _WIN32
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
@@ -172,19 +185,13 @@ namespace FEXCore::Context {
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
FEXCore::ForkableSharedMutex &GetCodeInvalidationMutex() override {
return CodeInvalidationMutex;
}
void MarkMemoryShared(FEXCore::Core::InternalThreadState *Thread) override;
void MarkMemoryShared() override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
// returns false if a handler was already registered
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr);
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) override;
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
@@ -193,7 +200,11 @@ namespace FEXCore::Context {
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
#ifdef JIT_X86_64
friend class FEXCore::CPU::X86JITCore;
#endif
friend class FEXCore::CPU::InterpreterCore;
friend class FEXCore::IR::Validation::IRValidation;
struct {
@@ -203,9 +214,6 @@ namespace FEXCore::Context {
// this is for internal use
bool ValidateIRarser { false };
// Used if the JIT needs to have its interrupt fault code emitted.
bool NeedsPendingInterruptFaultCheck { false };
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
@@ -223,6 +231,7 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
@@ -231,17 +240,25 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
} Config;
FEXCore::HostFeatures HostFeatures;
std::mutex ThreadCreationMutex;
FEXCore::Core::InternalThreadState* ParentThread{};
fextl::vector<FEXCore::Core::InternalThreadState*> Threads;
std::atomic_bool CoreShuttingDown{false};
bool NeedToCheckXID{true};
std::mutex IdleWaitMutex;
std::condition_variable IdleWaitCV;
std::atomic<uint32_t> IdleWaitRefCount{};
Event PauseWait;
bool Running{};
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
FEXCore::HostFeatures HostFeatures;
// CPUID depends on HostFeatures so needs to be initialized after that.
FEXCore::CPUIDEmu CPUID;
FEXCore::HLE::SyscallHandler *SyscallHandler{};
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
@@ -261,15 +278,25 @@ namespace FEXCore::Context {
ContextImpl();
~ContextImpl();
bool IsPaused() const { return !Running; }
void WaitForThreadsToRun();
void Stop(bool IgnoreCurrentThread);
void WaitForIdle();
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
void StartGdbServer();
void StopGdbServer();
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData *HostLink, const BlockDelinkerFunc &delinker);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, ExitFunctionLinkData *Record) {
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
return Fn(Frame, Record);
return Fn(Frame, record);
}
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
@@ -278,7 +305,7 @@ namespace FEXCore::Context {
auto Thread = Frame->Thread;
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
@@ -293,7 +320,7 @@ namespace FEXCore::Context {
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
struct CompileCodeResult {
void* CompiledCode;
@@ -304,8 +331,11 @@ namespace FEXCore::Context {
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
// same as CompileBlock, but aborts on failure
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
// Used for thread creation from syscalls
/**
@@ -317,6 +347,8 @@ namespace FEXCore::Context {
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
fextl::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
FEXCore::JITSymbols Symbols;
@@ -331,6 +363,10 @@ namespace FEXCore::Context {
}
}
void IncrementIdleRefCount() override {
++IdleWaitRefCount;
}
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
@@ -342,20 +378,6 @@ namespace FEXCore::Context {
UpdateAtomicTSOEmulationConfig();
}
// Returns if Software TSO emulation is required.
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
// This will still return true if on a single thread and TSO is currently disabled.
//
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
// we return consistent results.
//
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
bool SoftwareTSORequired() const {
if (SupportsHardwareTSO) return false;
return Config.TSOEnabled;
}
void EnableExitOnHLT() override { ExitOnHLT = true; }
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
@@ -363,6 +385,8 @@ namespace FEXCore::Context {
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
protected:
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
// If the hardware supports TSO then we don't need to emulate it through atomics.
@@ -375,6 +399,15 @@ namespace FEXCore::Context {
}
private:
/**
* @brief Does some final thread initialization
*
* @param Thread The internal FEX thread state object
*
* InitCore and CreateThread both call this to finish up thread object initialization
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Initializes the JIT compilers for the thread
*
@@ -384,8 +417,16 @@ namespace FEXCore::Context {
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
void WaitForIdleWithTimeout();
void NotifyPause();
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
// Entry Cache
std::mutex ExitMutex;
fextl::unique_ptr<GdbServer> DebugServer;
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
@@ -397,7 +438,9 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
std::shared_mutex CustomIRMutex;
std::atomic<bool> HasCustomIRHandlers{};
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void *, void *>> CustomIRHandlers;
FEXCore::CPU::DispatcherConfig DispatcherConfig;
};
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
}
@@ -1,19 +1,15 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "FEXCore/Core/X86Enums.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Context/Context.h"
#include "Interface/HLE/Thunks/Thunks.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <aarch64/cpu-aarch64.h>
#include <aarch64/instructions-aarch64.h>
#include <cpu-features.h>
@@ -29,9 +25,8 @@ namespace FEXCore::CPU {
// TODO: Allow x18 register allocation on Linux in the future to gain one more register.
namespace x64 {
#ifndef _M_ARM_64EC
// All but x19 and x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
@@ -39,23 +34,23 @@ namespace x64 {
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29,
// PF/AF must be last.
REG_PF, REG_AF,
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
};
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
// All these callee saved
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
FEXCore::ARMEmitter::Reg::r30,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
}};
// All are caller saved
@@ -71,150 +66,11 @@ namespace x64 {
};
// v8..v15 = (lower 64bits) Callee saved
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
// v0 ~ v1 are used as temps.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
// v0 ~ v3 are used as temps.
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
};
#else
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r0,
FEXCore::ARMEmitter::Reg::r1, FEXCore::ARMEmitter::Reg::r27,
// SP's register location isn't specified by the ARM64EC ABI, we choose to use r23
FEXCore::ARMEmitter::Reg::r23, FEXCore::ARMEmitter::Reg::r29,
FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r26,
FEXCore::ARMEmitter::Reg::r2, FEXCore::ARMEmitter::Reg::r3,
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r20,
FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22,
REG_PF, REG_AF,
};
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r14,FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r30,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
{FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7},
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
}};
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
};
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
};
#endif
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 7> PreserveAll_SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
};
constexpr uint32_t PreserveAll_SRAMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRA) {
switch (Reg.Idx()) {
case 0:
case 1:
case 2:
case 3:
case 4:
case 5:
case 6:
case 7:
case 8:
case 16:
case 17:
Mask |= (1U << Reg.Idx());
break;
default: break;
}
}
return Mask;
}()
};
// Dynamic GPRs
constexpr std::array<FEXCore::ARMEmitter::Register, 1> PreserveAll_Dynamic = {
// Only LR needs to get saved.
FEXCore::ARMEmitter::Reg::r30
};
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
// None.
};
constexpr uint32_t PreserveAll_SRAFPRMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPR) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs
// - v0-v7
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
// v0 ~ v1 are temps
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
};
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
// This is /all/ of the SRA registers
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> PreserveAll_SRAFPRSVE = SRAFPR;
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPRSVE) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs when the host supports SVE-256bit.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> PreserveAll_DynamicFPRSVE = {
// v0 ~ v1 are used as temps.
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
@@ -226,20 +82,19 @@ namespace x64 {
namespace x32 {
// All but x19 and x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 10> SRA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
// PF/AF must be last.
REG_PF, REG_AF,
};
constexpr std::array<FEXCore::ARMEmitter::Register, 15> RA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
// All these callee saved
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
// Registers only available on 32-bit
// All these are caller saved (except for r19).
@@ -251,10 +106,11 @@ namespace x32 {
FEXCore::ARMEmitter::Reg::r19,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 7> RAPair = {{
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
@@ -271,106 +127,11 @@ namespace x32 {
};
// v8..v15 = (lower 64bits) Callee saved
constexpr std::array<FEXCore::ARMEmitter::VRegister, 22> RAFPR = {
// v0 ~ v1 are used as temps.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 20> RAFPR = {
// v0 ~ v3 are used as temps.
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
};
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 5> PreserveAll_SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8,
};
constexpr uint32_t PreserveAll_SRAMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRA) {
switch (Reg.Idx()) {
case 0:
case 1:
case 2:
case 3:
case 4:
case 5:
case 6:
case 7:
case 8:
case 16:
case 17:
Mask |= (1U << Reg.Idx());
break;
default: break;
}
}
return Mask;
}()
};
// Dynamic GPRs
constexpr std::array<FEXCore::ARMEmitter::Register, 3> PreserveAll_Dynamic = {
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r30
};
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
// None.
};
constexpr uint32_t PreserveAll_SRAFPRMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPR) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs
// - v0-v7
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
// v0 ~ v1 are temps
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
};
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
// This is /all/ of the SRA registers
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> PreserveAll_SRAFPRSVE = SRAFPR;
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPRSVE) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs when the host supports SVE-256bit.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 22> PreserveAll_DynamicFPRSVE = {
// v0 ~ v1 are used as temps.
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
@@ -386,11 +147,11 @@ namespace x32 {
}
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr, size_t size)
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
: Emitter(size ? (uint8_t*)FEXCore::Allocator::VirtualAlloc(size, true) : nullptr, size)
, EmitterCTX {ctx}
#ifdef VIXL_SIMULATOR
, Simulator {&SimDecoder, stdout, vixl::aarch64::SimStack(SimulatorStackSize).Allocate()}
, Simulator {&SimDecoder}
#endif
{
#ifdef VIXL_SIMULATOR
@@ -403,10 +164,8 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
// Only setup the disassembler if enabled.
// vixl's decoder is expensive to setup.
if (Disassemble()) {
DisasmBuffer.resize(DISASM_BUFFER_SIZE);
Disasm = fextl::make_unique<vixl::aarch64::Disassembler>(DisasmBuffer.data(), DISASM_BUFFER_SIZE);
DisasmDecoder = fextl::make_unique<vixl::aarch64::Decoder>();
DisasmDecoder->AppendVisitor(Disasm.get());
DisasmDecoder->AppendVisitor(&Disasm);
}
#endif
@@ -419,12 +178,9 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
GeneralPairRegisters = x64::RAPair;
StaticFPRegisters = x64::SRAFPR;
GeneralFPRegisters = x64::RAFPR;
#ifdef _M_ARM_64EC
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
#endif
}
else {
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
StaticRegisters = x32::SRA;
GeneralRegisters = x32::RA;
@@ -435,6 +191,13 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
}
}
Arm64Emitter::~Arm64Emitter() {
auto BufferSize = GetBufferSize();
if (BufferSize) {
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
}
}
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
int Segments = Is64Bit ? 4 : 2;
@@ -455,15 +218,6 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
Segments = 2;
}
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
movn(s, Reg.W(), (~Constant) & 0xFFFF);
if (NOPPad) {
nop(); nop(); nop();
}
return;
}
int RequiredMoveSegments{};
// Count the number of move segments
@@ -639,37 +393,9 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
}
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
#ifndef VIXL_SIMULATOR
if (EmitterCTX->HostFeatures.SupportsAFP) {
// Disable AFP features when spilling registers.
//
// Disable FPCR.NEP and FPCR.AH
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
// Also interacts with RPRES to change reciprocal/rsqrt precision from 8-bit mantissa to 12-bit.
//
// Additional interesting AFP bits:
// FIZ(0): Flush Inputs to Zero
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
(1U << 2) | // NEP
(1U << 1)); // AH
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
if (!StaticRegisterAllocation()) {
return;
}
#endif
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
// is always static and almost certainly clobbered by the subsequent code.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
GPRSpillMask &= ~PFAFSpillMask;
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
auto Reg1 = StaticRegisters[i];
@@ -686,14 +412,6 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
}
}
// Now handle PF/AF
if (PFAFSpillMask) {
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
}
if (FPRs) {
if (EmitterCTX->HostFeatures.SupportsAVX) {
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
@@ -739,57 +457,21 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
}
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
[[maybe_unused]] bool FoundRegister{};
for (auto Reg : StaticRegisters) {
if (((1U << Reg.Idx()) & GPRFillMask)) {
TmpReg = Reg;
FoundRegister = true;
break;
}
if (!StaticRegisterAllocation()) {
return;
}
LOGMAN_THROW_A_FMT(FoundRegister, "Didn't have an SRA register to use as a temporary while spilling!");
#ifndef VIXL_SIMULATOR
if (EmitterCTX->HostFeatures.SupportsAFP) {
// Enable AFP features when filling JIT state.
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
// Enable FPCR.NEP and FPCR.AH
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
//
// Additional interesting AFP bits:
// FIZ(0): Flush Inputs to Zero
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
(1U << 2) | // NEP
(1U << 1)); // AH
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
}
#endif
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
// is always static and was almost certainly clobbered.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
if (FPRs) {
// Set up predicate registers.
// We don't bother spilling these in SpillStaticRegs,
// since all that matters is we restore them on a fill.
// It's not a concern if they get trounced by something else.
if (EmitterCTX->HostFeatures.SupportsSVE) {
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
}
if (EmitterCTX->HostFeatures.SupportsAVX) {
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
const auto Reg = StaticFPRegisters[i];
@@ -802,6 +484,8 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
if (GPRFillMask && FPRFillMask == ~0U) {
// Optimize the common case where we can fill four registers per instruction.
// Use one of the filling static registers before we fill it.
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
// Load the sse offset in to the temporary register
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
@@ -832,11 +516,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
}
}
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
GPRFillMask &= ~PFAFMask;
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i+1];
@@ -851,115 +530,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
}
}
// Now handle PF/AF
if (PFAFFillMask) {
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
}
}
void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
if (SVERegs) {
size_t i = 0;
for (; i < (VRegs.size() % 4); i += 2) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
st2b(Reg1.Z(), Reg2.Z(), PRED_TMP_32B, TmpReg, 0);
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 2);
}
for (; i < VRegs.size(); i += 4) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
const auto Reg3 = VRegs[i + 2];
const auto Reg4 = VRegs[i + 3];
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
}
}
else {
size_t i = 0;
for (; i < (VRegs.size() % 4); i += 2) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), TmpReg, 32);
}
for (; i < VRegs.size(); i += 4) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
const auto Reg3 = VRegs[i + 2];
const auto Reg4 = VRegs[i + 3];
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
}
}
}
void Arm64Emitter::PushGeneralRegisters(FEXCore::ARMEmitter::Register TmpReg, std::span<const FEXCore::ARMEmitter::Register> Regs) {
size_t i = 0;
for (; i < (Regs.size() % 2); ++i) {
const auto Reg1 = Regs[i];
str<ARMEmitter::IndexType::POST>(Reg1.X(), TmpReg, 16);
}
for (; i < Regs.size(); i += 2) {
const auto Reg1 = Regs[i];
const auto Reg2 = Regs[i + 1];
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
}
}
void Arm64Emitter::PopVectorRegisters(bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
if (SVERegs) {
size_t i = 0;
for (; i < (VRegs.size() % 4); i += 2) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
ld2b(Reg1.Z(), Reg2.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 2);
}
for (; i < VRegs.size(); i += 4) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
const auto Reg3 = VRegs[i + 2];
const auto Reg4 = VRegs[i + 3];
ld4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
}
} else {
size_t i = 0;
for (; i < (VRegs.size() % 4); i += 2) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), ARMEmitter::Reg::rsp, 32);
}
for (; i < VRegs.size(); i += 4) {
const auto Reg1 = VRegs[i];
const auto Reg2 = VRegs[i + 1];
const auto Reg3 = VRegs[i + 2];
const auto Reg4 = VRegs[i + 3];
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
}
}
}
void Arm64Emitter::PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Register> Regs) {
size_t i = 0;
for (; i < (Regs.size() % 2); ++i) {
const auto Reg1 = Regs[i];
ldr<ARMEmitter::IndexType::POST>(Reg1.X(), ARMEmitter::Reg::rsp, 16);
}
for (; i < Regs.size(); i += 2) {
const auto Reg1 = Regs[i];
const auto Reg2 = Regs[i + 1];
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
}
}
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
@@ -975,123 +545,64 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
// rsp capable move
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 2 == 0, "Needs to have multiple of 2 FPRs for RA");
if (CanUseSVE) {
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
const auto Reg1 = GeneralFPRegisters[i];
const auto Reg2 = GeneralFPRegisters[i + 1];
const auto Reg3 = GeneralFPRegisters[i + 2];
const auto Reg4 = GeneralFPRegisters[i + 3];
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
}
} else {
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
const auto Reg1 = GeneralFPRegisters[i];
const auto Reg2 = GeneralFPRegisters[i + 1];
const auto Reg3 = GeneralFPRegisters[i + 2];
const auto Reg4 = GeneralFPRegisters[i + 3];
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
}
}
// Push the vector registers
PushVectorRegisters(TmpReg, CanUseSVE, GeneralFPRegisters);
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
}
// Push the general registers.
PushGeneralRegisters(TmpReg, ConfiguredDynamicRegisterBase);
#ifndef _M_ARM_64EC
str(ARMEmitter::XReg::lr, TmpReg, 0);
#endif
}
void Arm64Emitter::PopDynamicRegsAndLR() {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
// Pop vectors first
PopVectorRegisters(CanUseSVE, GeneralFPRegisters);
if (CanUseSVE) {
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
const auto Reg1 = GeneralFPRegisters[i];
const auto Reg2 = GeneralFPRegisters[i + 1];
const auto Reg3 = GeneralFPRegisters[i + 2];
const auto Reg4 = GeneralFPRegisters[i + 3];
ld4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
}
} else {
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
const auto Reg1 = GeneralFPRegisters[i];
const auto Reg2 = GeneralFPRegisters[i + 1];
const auto Reg3 = GeneralFPRegisters[i + 2];
const auto Reg4 = GeneralFPRegisters[i + 3];
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
}
}
// Pop GPRs second
PopGeneralRegisters(ConfiguredDynamicRegisterBase);
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
}
#ifndef _M_ARM_64EC
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
#endif
}
void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs) {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
: Core::CPUState::XMM_SSE_REG_SIZE;
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs{};
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs{};
uint32_t PreserveSRAMask{};
uint32_t PreserveSRAFPRMask{};
if (EmitterCTX->Config.Is64BitMode()) {
DynamicGPRs = x64::PreserveAll_Dynamic;
DynamicFPRs = x64::PreserveAll_DynamicFPR;
PreserveSRAMask = x64::PreserveAll_SRAMask;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRMask;
if (CanUseSVE) {
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
}
}
else {
DynamicGPRs = x32::PreserveAll_Dynamic;
DynamicFPRs = x32::PreserveAll_DynamicFPR;
PreserveSRAMask = x32::PreserveAll_SRAMask;
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRMask;
if (CanUseSVE) {
DynamicFPRs = x32::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRSVEMask;
}
}
const auto GPRSize = AlignUp(DynamicGPRs.size(), 2) * Core::CPUState::GPR_REG_SIZE;
const auto FPRSize = DynamicFPRs.size() * FPRRegSize;
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
// Spill the static registers.
SpillStaticRegs(TmpReg, true, PreserveSRAMask, PreserveSRAFPRMask);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
// rsp capable move
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
// Push the vector registers.
PushVectorRegisters(TmpReg, CanUseSVE, DynamicFPRs);
// Push the general registers.
PushGeneralRegisters(TmpReg, DynamicGPRs);
}
void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs{};
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs{};
uint32_t PreserveSRAMask{};
uint32_t PreserveSRAFPRMask{};
if (EmitterCTX->Config.Is64BitMode()) {
DynamicGPRs = x64::PreserveAll_Dynamic;
DynamicFPRs = x64::PreserveAll_DynamicFPR;
PreserveSRAMask = x64::PreserveAll_SRAMask;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRMask;
if (CanUseSVE) {
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
}
}
else {
DynamicGPRs = x32::PreserveAll_Dynamic;
DynamicFPRs = x32::PreserveAll_DynamicFPR;
PreserveSRAMask = x32::PreserveAll_SRAMask;
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRMask;
if (CanUseSVE) {
DynamicFPRs = x32::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRSVEMask;
}
}
// Fill the static registers.
FillStaticRegs(true, PreserveSRAMask, PreserveSRAFPRMask);
// Pop the vector registers.
PopVectorRegisters(CanUseSVE, DynamicFPRs);
// Pop the general registers.
PopGeneralRegisters(DynamicGPRs);
}
void Arm64Emitter::Align16B() {
@@ -1,10 +1,10 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "FEXCore/Utils/EnumUtils.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/ObjectCache/Relocations.h"
#include <aarch64/assembler-aarch64.h>
@@ -21,7 +21,6 @@
#endif
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/vector.h>
#include <array>
#include <cstddef>
@@ -29,45 +28,22 @@
#include <utility>
#include <span>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::CPU {
// Contains the address to the currently available CPU state
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
#ifndef _M_ARM_64EC
// GPR temporaries. Only x3 can be used across spill boundaries
// so if these ever need to change, be very careful about that.
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x0;
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x1;
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x2;
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x3;
constexpr bool TMP_ABIARGS = true;
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
// Vector temporaries
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v0;
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
#else
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x10;
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x11;
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x12;
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x13;
constexpr bool TMP_ABIARGS = false;
// We pin r11/r12 as PF/AF respectively for arm64ec, as r26/r27 are used for SRA.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r9;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r24;
// Vector temporaries
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v16;
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v17;
#endif
constexpr auto VTMP3 = FEXCore::ARMEmitter::VReg::v2;
constexpr auto VTMP4 = FEXCore::ARMEmitter::VReg::v3;
// Predicate register temporaries (used when AVX support is enabled)
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
@@ -75,12 +51,12 @@ constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v17;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
protected:
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr = nullptr, size_t size = 0);
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size);
~Arm64Emitter();
FEXCore::Context::ContextImpl *EmitterCTX;
vixl::aarch64::CPU CPU;
@@ -123,52 +99,12 @@ protected:
// We can't guarantee only the lower 64bits are used so flush everything
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
// Generic push and pop vector registers.
void PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs);
void PushGeneralRegisters(FEXCore::ARMEmitter::Register TmpReg, std::span<const FEXCore::ARMEmitter::Register> Regs);
void PopVectorRegisters(bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs);
void PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Register> Regs);
void PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg);
void PopDynamicRegsAndLR();
void PushCalleeSavedRegisters();
void PopCalleeSavedRegisters();
// Spills and fills SRA/Dynamic registers that are required for Arm64 `preserve_all` ABI.
// This ABI changes most registers to be callee saved.
// Caller Saved:
// - X0-X8, X16-X18.
// - v0-v7
// - For 256-bit SVE hosts: top 128-bits of v8-v31
//
// Callee Saved:
// - X9-X15, X19-X31
// - Low 128-bits of v8-v31
void SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true);
void FillForPreserveAllABICall(bool FPRs = true);
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
if (SupportsPreserveAllABI) {
SpillForPreserveAllABICall(TmpReg, FPRs);
}
else {
SpillStaticRegs(TmpReg, FPRs);
PushDynamicRegsAndLR(TmpReg);
}
}
void FillForABICall(bool SupportsPreserveAllABI, bool FPRs = true) {
if (SupportsPreserveAllABI) {
FillForPreserveAllABICall(FPRs);
}
else {
PopDynamicRegsAndLR();
FillStaticRegs(FPRs);
}
}
void Align16B();
#ifdef VIXL_SIMULATOR
@@ -235,31 +171,21 @@ protected:
// Call type
dc32(vixl::aarch64::kCallRuntime);
}
#else
template<typename R, typename... P>
void GenerateRuntimeCall(R (*Function)(P...)) {
// Explicitly doing nothing.
}
template<typename R, typename... P>
void GenerateIndirectRuntimeCall(ARMEmitter::Register Reg) {
// Explicitly doing nothing.
}
#endif
#ifdef VIXL_SIMULATOR
vixl::aarch64::Decoder SimDecoder;
vixl::aarch64::Simulator Simulator;
constexpr static size_t SimulatorStackSize = 8 * 1024 * 1024;
#endif
#ifdef VIXL_DISASSEMBLER
fextl::vector<char> DisasmBuffer;
constexpr static int DISASM_BUFFER_SIZE {256};
fextl::unique_ptr<vixl::aarch64::Disassembler> Disasm;
vixl::aarch64::Disassembler Disasm;
fextl::unique_ptr<vixl::aarch64::Decoder> DisasmDecoder;
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
#endif
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
};
}
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* ALU instruction emitters.
*
* Almost all of these operations have `ARMEmitter::Size` as their first argument.
@@ -35,10 +34,8 @@ public:
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void adr(FEXCore::ARMEmitter::Register rd, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADR });
void adr(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADR });
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
}
@@ -64,10 +61,8 @@ public:
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void adrp(FEXCore::ARMEmitter::Register rd, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADRP });
void adrp(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADRP });
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
}
@@ -109,7 +104,7 @@ public:
}
}
void LongAddressGen(FEXCore::ARMEmitter::Register rd, ForwardLabel* Label) {
Label->Insts.emplace_back(SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN });
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN });
// Emit a register index and a nop. These will be backpatched.
dc32(rd.Idx());
nop();
@@ -770,21 +765,6 @@ public:
EvaluateIntoFlags(Op, 1, rn);
}
void cfinv() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0001'1111;
dc32(Op);
}
void axflag() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0101'1111;
dc32(Op);
}
void xaflag() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0011'1111;
dc32(Op);
}
// Conditional compare - register
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0011'1010'010 << 21;
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* ASIMD instruction emitters.
*
* This contains emitters for vector operations explicitly.
@@ -60,7 +59,7 @@ public:
}
void sha256su1(FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn, FEXCore::ARMEmitter::VRegister rm) {
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
Crypto3RegSHA(Op, 0b110, rd, rn, rm);
Crypto3RegSHA(Op, 0b100, rd, rn, rm);
}
// Cryptographic two-register SHA
@@ -1,4 +1,3 @@
// SPDX-License-Identifier: MIT
/* Branch instruction emitters.
*
* Most of these instructions will use `BackwardLabel`, `ForwardLabel`, or `BiDirectionLabel` to determine where a branch targets.
@@ -18,10 +17,8 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void b(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
}
@@ -47,10 +44,8 @@ public:
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bc(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void bc(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
}
@@ -106,10 +101,8 @@ public:
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
void b(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
@@ -137,10 +130,8 @@ public:
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bl(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
void bl(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
@@ -171,10 +162,8 @@ public:
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0100 << 24;
@@ -205,10 +194,8 @@ public:
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0101 << 24;
@@ -238,11 +225,8 @@ public:
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0110 << 24;
@@ -271,11 +255,8 @@ public:
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
Loaded 100 of 879 files, more files were not shown because too many files have changed in this diff. Show more