Compare commits

..
3 Commits
Author SHA1 Message Date
Ryan Houdek ea20429351 Docs: Update for release FEX-2507.1 2025-07-11 11:37:44 -07:00
Alyssa Rosenzweig 91828efa7a JIT: fix divisor masking
oversight. should fix Steam.

Fixes: de4becc26 ("OpcodeDispatcher: mask certain divisors")
Closes: #4652
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2025-07-11 11:34:45 -07:00
Billy Laws cce605d5e0 PoolBufferWithTimedRetirement: Unclaim in dtor
Buffers are tied to the lifetime of their owned flag, and as that
is a member of PoolBufferWithTimedRetirement we must always unclaim here.

Avoids the need to manually remember this quirk (which was forgot for the
temporary compilation buffer in JIT.cpp) at every use-site.
2025-07-11 11:34:07 -07:00
1241 changed files with 80574 additions and 111245 deletions

No files matched your search

+2 -2
View File
@@ -32,7 +32,7 @@ AttributeMacros:
BinPackArguments: true
BinPackParameters: true
BitFieldColonSpacing: Both
BreakAfterAttributes: Leave
BreakAfterAttributes: Always # clang 16 required
BreakBeforeBraces: Attach
BreakBeforeBinaryOperators: None
BreakBeforeInlineASMColon: OnlyMultiline # clang 16 required
@@ -60,7 +60,7 @@ IndentRequires: false
IndentWidth: 2
InsertBraces: true
KeepEmptyLinesAtTheStartOfBlocks: true
LambdaBodyIndentation: Signature
LambdaBodyIndentation: OuterScope
LineEnding: LF # clang 16 required
MaxEmptyLinesToKeep: 2
NamespaceIndentation: Inner
+5 -4
View File
@@ -1,12 +1,13 @@
# This file is used to ignore files and directories from clang-format
# Ignore all files in the External directory
External/*
Source/Common/cpp-optparse/*
# Files with human-indented tables for readability - don't mess with these
FEXCore/Source/Interface/Core/X86Tables/*.cpp
FEXCore/Source/Interface/Core/X86Tables/*
# Inline headers with list-like content that can't be processed individually
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
# Include files in unittests
unittests/*ASM/Includes/*.inc
-9
View File
@@ -16,12 +16,3 @@
# Reformat of CodeEmitter inl files
8760c593ece92d7e9fa94c40da0368fd367c9cad
# Whole-tree reformat with clang-format-19
5267cde60e7642852d18f20ae8568643bb5293d5
# Minor reformat with clang-format-19
9fdd96af61c969cb5732471223f00eda64b7a069
# Reformat of X86Tables.h
ba2b0ef809f66f1a6d334f000798fa2ceafab26f
+185 -88
View File
@@ -13,7 +13,6 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_PORTABLE: 1
jobs:
build_plus_test:
@@ -24,138 +23,236 @@ jobs:
fail-fast: false
steps:
- uses: actions/checkout@v6
with:
fetch-depth: '0'
fetch-tags: 'true'
- uses: actions/checkout@v3
- name: Set runner info
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
echo "runner_name=$(hostname)" >> $GITHUB_ENV
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Setup Build Environment
uses: ./.github/workflows/setup-env
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
run: |
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True \
-DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True \
-DCMAKE_INSTALL_PREFIX="$PWD"/build/install
# These steps make a lot of noise but rarely fail.
# Put them in a separate step to make normal build logs easier to parse
- name: Noisy Build Targets
run: cmake --build build --target asm_files 32bit_asm_files JemallocLibs Catch2 vixl cephes_128bit
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
run: cmake --build build
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE
- name: Install
run: cmake --build build --target install
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
# GCC tests
- name: GCC64 Target Tests
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_64
- name: GCC64 Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: gcc_target_tests_64
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
- name: GCC32 Target Tests
- name: gcc target tests 32
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
- name: GCC32 Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: gcc_target_tests_32
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
# API tests
- name: API Tests
- name: APITest tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target api_tests
- name: APITest Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: api_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
- name: FEXCore API Tests
- name: FEXCore APITest tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
- name: FEXCore APITest Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: fexcore_apitests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
# ARM emission tests
- name: ARM Emitter Tests
- name: ARMEmitter tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target emitter_tests
- name: ARMEmitter Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: emitter_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ARMEmitterTests.log || true
# Linux tests
- name: FEX Linux Tests
- name: FEXLinuxTests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
- name: FEXLinuxTests Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: fex_linux_tests_all
env:
FEX_PORTABLE: 0
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
# Thunking
- name: Thunkgen tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target thunkgen_tests
- name: Thunkgen Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: thunkgen_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
- name: Test GL No-Thunks
if: ${{ always() && matrix.arch[1] == 'x64' }}
uses: ./.github/workflows/test
with:
target: thunk_functional_tests_nothunks
if: matrix.arch[1] == 'x64'
working-directory: ${{runner.workspace}}/build
shell: bash
env:
DISPLAY: ':0'
DISPLAY: ":0"
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_nothunks
- name: No thunks Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_NoThunkResults.log || true
- name: Test GL Thunks
if: ${{ always() && matrix.arch[1] == 'x64' }}
uses: ./.github/workflows/test
with:
target: thunk_functional_tests_thunks
if: matrix.arch[1] == 'x64'
working-directory: ${{runner.workspace}}/build
shell: bash
env:
DISPLAY: ':0'
DISPLAY: ":0"
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_thunks
- name: Thunks Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkResults.log || true
# ASM tests
- name: ASM Tests
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: asm_tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
# POSIX tests
- name: POSIX Tests
- name: ASM Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: posix_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
# GVisor tests
- name: GVisor Tests
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: gvisor_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: gvisor tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gvisor_tests
- name: GVisor Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GVisor.log || true
# Struct verifier tests
- name: Struct verifier tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target struct_verifier
- name: Struct verifier Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: struct_verifier
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
run: cmake --build build --target remove_old_shm_regions
shell: bash
working-directory: ${{runner.workspace}}/build
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: actions/upload-artifact@v6
uses: 'actions/upload-artifact@v4'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
path: results/*.log
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
+128 -59
View File
@@ -20,7 +20,6 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_PORTABLE: 1
jobs:
glibc_fault_test:
@@ -31,94 +30,164 @@ jobs:
fail-fast: false
steps:
- uses: actions/checkout@v6
with:
fetch-depth: '0'
fetch-tags: 'true'
- uses: actions/checkout@v3
- name: Set runner info
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
echo "runner_name=$(hostname)" >> $GITHUB_ENV
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Setup Build Environment
uses: ./.github/workflows/setup-env
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
run: |
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False \
-DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True \
-DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False \
-DCMAKE_INSTALL_PREFIX="$PWD"/build/install
# These steps make a lot of noise but rarely fail.
# Put them in a separate step to make normal build logs easier to parse
- name: Noisy Build Targets
run: cmake --build build --target asm_files 32bit_asm_files JemallocLibs Catch2 vixl cephes_128bit
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
run: cmake --build build
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE
- name: Install
run: cmake --build build --target install
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
# GCC tests
- name: GCC64 Target Tests
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_64
- name: GCC64 Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: gcc_target_tests_64
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
- name: GCC32 Target Tests
- name: gcc target tests 32
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
- name: GCC32 Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: gcc_target_tests_32
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
# API Tests
- name: API Tests
- name: APITest tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target api_tests
- name: APITest Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: api_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
- name: FEXCore API Tests
- name: FEXCore APITest tests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
- name: FEXCore APITest Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: fexcore_apitests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
# Linux tests
- name: FEX Linux Tests
- name: FEXLinuxTests
working-directory: ${{runner.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
- name: FEXLinuxTests Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: fex_linux_tests_all
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
# ASM Tests
- name: ASM Tests
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: asm_tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
# POSIX Tests
- name: POSIX Tests
- name: ASM Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: posix_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: Posix Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the posixtest
run: cmake --build . --config $BUILD_TYPE --target posix_tests
- name: Posix Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
run: cmake --build build --target remove_old_shm_regions
shell: bash
working-directory: ${{runner.workspace}}/build
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: actions/upload-artifact@v6
uses: 'actions/upload-artifact@v4'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
path: results/*.log
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
+64 -26
View File
@@ -13,7 +13,6 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_PORTABLE: 1
jobs:
hostrunner_tests:
@@ -24,45 +23,84 @@ jobs:
fail-fast: false
steps:
- uses: actions/checkout@v6
with:
fetch-depth: '0'
fetch-tags: 'true'
- uses: actions/checkout@v3
- name: Set runner info
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
echo "runner_name=$(hostname)" >> $GITHUB_ENV
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Setup Build Environment
uses: ./.github/workflows/setup-env
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
run: |
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False \
-DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
# These steps make a lot of noise but rarely fail.
# Put them in a separate step to make normal build logs easier to parse
- name: Noisy Build Targets
run: cmake --build build --target asm_files 32bit_asm_files JemallocLibs Catch2 vixl cephes_128bit
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
run: cmake --build build
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE
# ASM tests
- name: ASM Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: asm_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: actions/upload-artifact@v6
uses: 'actions/upload-artifact@v4'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
path: results/*.log
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
+96 -28
View File
@@ -23,56 +23,124 @@ jobs:
fail-fast: false
steps:
- uses: actions/checkout@v6
with:
fetch-depth: '0'
fetch-tags: 'true'
- uses: actions/checkout@v3
- name: Set runner info
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
echo "runner_name=$(hostname)" >> $GITHUB_ENV
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Setup Build Environment
uses: ./.github/workflows/setup-env
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name: Set VIXL_SIM_ENABLED
- name : submodule checkout
# Need to update submodules
run: |
case '${{ matrix.arch[1] }}' in
x64) _sim=True ;;
ARM64) _sim=False ;;
esac
echo "VIXL_SIM_ENABLED=$_sim" >> $GITHUB_ENV
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Set vixl_sim x86
if: matrix.arch[1] == 'x64'
run: |
echo "VIXL_SIM_ENABLED=True" >> $GITHUB_ENV
- name: Set vixl_sim Arm64
if: matrix.arch[1] == 'ARM64'
run: |
echo "VIXL_SIM_ENABLED=False" >> $GITHUB_ENV
- name: Configure CMake
run: |
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED \
-DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
shell: bash
env:
FEX_DISABLETELEMETRY: 1
run: cmake --build build --target CodeSizeValidation instcountci_test_files
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE --target CodeSizeValidation instcountci_test_files
- name: Instruction Count Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target instcountci_tests
- name: Instruction Count Test Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: instcountci_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_InstCountCI.log || true
- name: Update local repo instcount
if: ${{ always() }}
run: cmake --build build --target instcountci_update_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: cmake --build . --config $BUILD_TYPE --target instcountci_update_tests
- name: Check InstCountCI diff
- name: Get instcountCI diff
if: ${{ always() }}
run: git --no-pager diff --exit-code HEAD
shell: bash
working-directory: ${{github.workspace}}/
run: git diff --output=${{runner.workspace}}/build/InstCountCI.diff
- name: Check if InstCountCI Diff exists
if: ${{ always() }}
shell: bash
working-directory: ${{github.workspace}}/
# Check if the file is empty
run: sh -c "! test -s ${{runner.workspace}}/build/InstCountCI.diff"
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: actions/upload-artifact@v6
uses: 'actions/upload-artifact@v4'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
path: results/*.log
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
- name: Upload results InstCountCI
if: ${{ always() }}
uses: 'actions/upload-artifact@v4'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}-instcountci
path: ${{runner.workspace}}/build/InstCountCI.diff
retention-days: 3
+66 -18
View File
@@ -20,10 +20,7 @@ jobs:
fail-fast: false
steps:
- uses: actions/checkout@v6
with:
fetch-depth: '0'
fetch-tags: 'true'
- uses: actions/checkout@v3
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
@@ -31,23 +28,74 @@ jobs:
- name: Add MingGW to PATH
run: echo "$HOME/llvm-mingw/build/bin/" >> $GITHUB_PATH
- name: Set CC
- name: Set CC x86
if: matrix.arch[1] == 'x64'
run: |
case '${{ matrix.arch[1] }}' in
x64) _cpu=x86_64 ;;
ARM64) _cpu=aarch64 ;;
ARM64EC) _cpu=arm64ec ;;
esac
echo "MINGW_TRIPLE=${_cpu}-w64-mingw32" >> $GITHUB_ENV
echo "MINGW_TRIPLE=x86_64-w64-mingw32" >> $GITHUB_ENV
- name: Setup Build Environment
uses: ./.github/workflows/setup-env
- name: Set CC Arm64
if: matrix.arch[1] == 'ARM64'
run: |
echo "MINGW_TRIPLE=aarch64-w64-mingw32" >> $GITHUB_ENV
- name: Set CC Arm64EC
if: matrix.arch[1] == 'ARM64EC'
run: |
echo "MINGW_TRIPLE=arm64ec-w64-mingw32" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
run: |
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake \
-DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False \
-DCMAKE_INSTALL_PREFIX="$PWD"/build/install
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
run: cmake --build build
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: 'actions/upload-artifact@v4'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
+31 -18
View File
@@ -1,7 +1,7 @@
# Inspired by LLVM's pr-code-format.yml at
# Inspired by LLVM's pr-code-format.yml at
# https://github.com/llvm/llvm-project/blob/main/.github/workflows/pr-code-format.yml
name: Check code formatting
name: "Check code formatting"
on:
pull_request:
branches:
@@ -13,7 +13,7 @@ jobs:
if: github.repository == 'FEX-Emu/FEX'
steps:
- name: Checkout
- name: Fetch FEX sources
uses: actions/checkout@v4
with:
ref: ${{ github.event.pull_request.head.sha }}
@@ -27,37 +27,50 @@ jobs:
deepen_length: 500
- name: Get changed files
id: changed-files
uses: step-security/changed-files@3dbe17c78367e7d60f00d78ae6781a35be47b4a1 # v45.0.1
with:
separator: ","
skip_initial_fetch: true
- name: "Listed files"
env:
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
run: |
BASE=$(git merge-base main HEAD)
FILES=$(git diff --name-only "$BASE" | tr '\n' ',' | sed 's/,$//')
echo "CHANGED_FILES=$FILES" >> $GITHUB_ENV
echo "Formatting files:"
echo "$CHANGED_FILES"
echo "Changed files:"
echo "$FILES"
- name: Check for correct clang-format version
run: clang-format --version | grep -qF '16.0.6'
- name: Check git-clang-format-19 exists
run: which git-clang-format-19
- name: Check git-clang-format-16 exists
run: which git-clang-format-16
- name: Setup Python env
uses: actions/setup-python@v4
with:
python-version: 3.11
cache: pip
cache-dependency-path: ./External/code-format-helper/requirements_formatting.txt
python-version: '3.11'
cache: 'pip'
cache-dependency-path: './External/code-format-helper/requirements_formatting.txt'
- name: Install python dependencies
run: pip install -r ./External/code-format-helper/requirements_formatting.txt
- name: Run code formatter
env:
CLANG_FORMAT_PATH: git-clang-format-19
CLANG_FORMAT_PATH: 'git-clang-format-16'
GITHUB_PR_NUMBER: ${{ github.event.pull_request.number }}
START_REV: ${{ github.event.pull_request.base.sha }}
END_REV: ${{ github.event.pull_request.head.sha }}
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
# TODO(pmatos): Once we adopt v18, we should be able
# to take advantage of the new --diff_from_common_commit option
# explicitly in code-format-helper.py and not have to diff starting at
# the merge base.
run: |
python ./External/code-format-helper/code-format-helper.py \
--repo "FEX-Emu/FEX" \
--issue-number "$GITHUB_PR_NUMBER" \
--start-rev "$START_REV" \
--end-rev "$END_REV" \
--repo "FEX-emu/FEX" \
--issue-number $GITHUB_PR_NUMBER \
--start-rev $(git merge-base $START_REV $END_REV) \
--end-rev $END_REV \
--changed-files "$CHANGED_FILES"
-33
View File
@@ -1,33 +0,0 @@
name: Setup Build Environment
description: Setup RootFS and build environment
inputs:
setup-rootfs:
description: 'Whether or not to set up the rootfs'
default: true
runs:
using: composite
steps:
- name: Set rootfs paths
if: ${{ inputs.setup-rootfs == 'true' }}
shell: bash
run: |
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
if: ${{ inputs.setup-rootfs == 'true' }}
shell: bash
run: python3 Scripts/CI_FetchRootFS.py
- name: Checkout Submodules
shell: bash
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
shell: bash
run: rm -Rf build
-72
View File
@@ -1,72 +0,0 @@
name: steamrt4 build
on:
push:
branches:
- main
pull_request:
branches:
- main
env:
DEBIAN_FRONTEND: noninteractive
BUILD_TYPE: Release
CC: clang
CXX: clang++
jobs:
steamrt4_build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARM64, distrobox]]
fail-fast: false
steps:
- uses: actions/checkout@v6
with:
fetch-depth: '0'
fetch-tags: 'true'
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Setup Build Environment
uses: ./.github/workflows/setup-env
with:
setup-rootfs: false
# Setup everything required.
- name : distrobox setup
run: |
distrobox create -Y -i registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306 steamrt4 || true
distrobox upgrade steamrt4
distrobox enter --name steamrt4 -- sudo apt-get install -y \
git cmake ninja-build ccache \
lld clang \
libclang-dev llvm-dev \
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross \
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
- name: Configure CMake
run: |
distrobox enter --name steamrt4 -- cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE \
-G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True \
-DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld \
-DCMAKE_INSTALL_PREFIX=/usr
- name: Build
run: distrobox enter --name steamrt4 -- cmake --build build
- name: install
run: DESTDIR="$PWD"/install distrobox enter --name steamrt4 -- cmake --build build -t install
- name: Upload libraries
uses: actions/upload-artifact@v6
timeout-minutes: 1
with:
overwrite: true
name: steamrt4_steampipe_depot
path: ${{ github.workspace }}/install/*
retention-days: 60
compression-level: 9
-21
View File
@@ -1,21 +0,0 @@
name: Run Test and Store Logs
description: Run a test and store the log.
inputs:
target:
description: 'The test target to run'
required: true
runs:
using: composite
steps:
- name: Run Tests
shell: bash
run: cmake --build build --target ${{ inputs.target }}
- name: Move and Truncate Results
if: ${{ always() }}
shell: bash
run: |
mkdir -p results
mv build/Testing/Temporary/LastTest.log results/${{ inputs.target }}.log || true
truncate --size="<20M" results/${{ inputs.target }}.log || true
+85 -33
View File
@@ -13,7 +13,6 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_PORTABLE: 1
jobs:
vixl_simulator:
@@ -25,59 +24,112 @@ jobs:
fail-fast: false
steps:
- uses: actions/checkout@v6
with:
fetch-depth: '0'
fetch-tags: 'true'
- uses: actions/checkout@v3
- name: Set runner info
- name: Set runner label
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
- name: Set rootfs paths
run: |
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
echo "runner_name=$(hostname)" >> $GITHUB_ENV
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Setup Build Environment
uses: ./.github/workflows/setup-env
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean Build Environment
run: rm -Rf ${{runner.workspace}}/build
- name: Create Build Environment
# Some projects don't allow in-source building, so create a separate build directory
# We'll use this as our working directory for all subsequent commands
run: cmake -E make_directory ${{runner.workspace}}/build
- name: Configure CMake
run: |
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_LTO=False \
-DENABLE_VIXL_DISASSEMBLER=True -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
# These steps make a lot of noise but rarely fail.
# Put them in a separate step to make normal build logs easier to parse
- name: Noisy Build Targets
run: cmake --build build --target asm_files 32bit_asm_files JemallocLibs Catch2 vixl cephes_128bit
# Use a bash shell so we can use the same syntax for environment variable
# access regardless of the host operating system
shell: bash
working-directory: ${{runner.workspace}}/build
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
run: cmake --build build
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the build. You can specify a specific target with "--target <NAME>"
run: cmake --build . --config $BUILD_TYPE
- name: ASM Tests - SVE256
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test SVE256 Results move
if: ${{ always() }}
uses: ./.github/workflows/test
with:
target: asm_tests
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM_SVE256Bit.log || true
- name: ASM Tests - SVE128
if: ${{ always() }}
uses: ./.github/workflows/test
working-directory: ${{runner.workspace}}/build
shell: bash
env:
FEX_FORCESVEWIDTH: "128"
with:
target: asm_tests
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test 128-bit Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM_SVE128Bit.log || true
- name: ASM Tests - ASIMD
if: ${{ always() }}
uses: ./.github/workflows/test
working-directory: ${{runner.workspace}}/build
shell: bash
env:
FEX_HOSTFEATURES: "disablesve"
with:
target: asm_tests
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target asm_tests
- name: ASM Test ASIMD Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM_ASIMD.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
- name: Upload results
if: ${{ always() }}
uses: actions/upload-artifact@v6
uses: 'actions/upload-artifact@v4'
timeout-minutes: 1
with:
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
path: results/*.log
name: Results-${{ env.runner_name }}
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
retention-days: 3
-35
View File
@@ -1,35 +0,0 @@
name: Wine DLL Build
description: Build a wow64 or arm64ec Wine DLL
inputs:
target:
description: 'The target (arm64ec or wow64)'
required: true
runs:
using: composite
steps:
- name: Clean Build Environment
shell: bash
run: rm -Rf build_${{ inputs.target }}
- name: Configure CMake
shell: bash
run: |
case "${{ inputs.target }}" in
wow64) _cc=aarch64 ;;
arm64ec) _cc=arm64ec ;;
esac
cmake -S . -B build_${{ inputs.target }} -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=Data/CMake/toolchain_mingw.cmake \
-DMINGW_TRIPLE=${_cc}-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja \
-DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False \
-DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=generic -DTUNE_CPU=none
- name: Build
shell: bash
run: cmake --build build_${{ inputs.target }}
- name: Install
shell: bash
run: DESTDIR="$PWD"/install cmake --build build_${{ inputs.target }} -t install
-55
View File
@@ -1,55 +0,0 @@
name: Wine DLL artifacts
on:
push:
branches:
- main
env:
BUILD_TYPE: Release
jobs:
wine_dll_artifacts:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARM64, mingw]]
fail-fast: false
steps:
- uses: actions/checkout@v6
with:
fetch-depth: '0'
fetch-tags: 'true'
- name: Add MingGW to PATH
run: echo "$HOME/llvm-mingw/build/bin/" >> $GITHUB_PATH
- name: Checkout Submodules
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean install directory
run: rm -Rf install
- name: Build (wow64)
uses: ./.github/workflows/wine_build
with:
target: wow64
- name: Build (arm64ec)
uses: ./.github/workflows/wine_build
with:
target: arm64ec
- name: Upload libraries
uses: actions/upload-artifact@v6
timeout-minutes: 1
with:
overwrite: true
name: wine_dll_artifacts
path: ${{ github.workspace }}/install/usr/lib/wine/aarch64-windows/lib*.dll
retention-days: 60
compression-level: 9
-2
View File
@@ -11,5 +11,3 @@ out/
.vs/
*.pyc
.cache
.idea/
CMakeLists.txt.user
-32
View File
@@ -1,32 +0,0 @@
variables:
DEBIAN_FRONTEND: noninteractive
GIT_SUBMODULE_STRATEGY: recursive
GIT_DEPTH: 0
CC: clang
CXX: clang++
aarch64:
image: registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306
tags:
- docker
- linux
- arm64
- aarch64
script:
- apt-get -y update
- apt-get install -y
git cmake ninja-build ccache
lld clang
libclang-dev llvm-dev
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
- cmake -E make_directory build/
- cmake -DCMAKE_BUILD_TYPE=Release -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=armv8.2-a -DTUNE_CPU=none . -B build/
- cmake --build build/ --config Release
- DESTDIR=$(pwd)/install/ cmake --build build/ --config Release -t install
artifacts:
name: "steamrt artifacts"
untracked: false
paths:
- install/
+7 -13
View File
@@ -17,6 +17,9 @@
shallow = true
path = External/fex-gcc-target-tests-bins
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
[submodule "External/jemalloc"]
path = External/jemalloc
url = https://github.com/FEX-Emu/jemalloc.git
[submodule "External/fmt"]
path = External/fmt
url = https://github.com/fmtlib/fmt.git
@@ -29,6 +32,10 @@
[submodule "External/Catch2"]
path = External/Catch2
url = https://github.com/catchorg/Catch2.git
[submodule "External/robin-map"]
shallow = true
path = External/robin-map
url = https://github.com/FEX-Emu/robin-map.git
[submodule "External/Vulkan-Headers"]
shallow = true
path = External/Vulkan-Headers
@@ -39,16 +46,3 @@
[submodule "External/tracy"]
path = External/tracy
url = https://github.com/wolfpld/tracy
[submodule "External/range-v3"]
path = External/range-v3
url = https://github.com/ericniebler/range-v3.git
[submodule "External/zydis"]
shallow = true
path = External/zydis
url = https://github.com/zyantific/zydis.git
[submodule "External/unordered_dense"]
path = External/unordered_dense
url = https://github.com/martinus/unordered_dense.git
[submodule "External/rpmalloc"]
path = External/rpmalloc
url = https://github.com/FEX-Emu/rpmalloc.git
+224 -278
View File
@@ -1,146 +1,85 @@
cmake_minimum_required(VERSION 3.14)
project(FEX C CXX ASM)
include(CheckIncludeFiles)
check_include_files("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
INCLUDE (CheckIncludeFiles)
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests (requires x86 compiler)" FALSE)
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
option(BUILD_THUNKS "Build thunks" FALSE)
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
option(ENABLE_CLANG_THUNKS "Build thunks with clang" TRUE)
option(ENABLE_IWYU "Enable the Include What You Use sanitizer" FALSE)
option(ENABLE_IWYU "Enables include what you use program" FALSE)
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
set(USE_LINKER "" CACHE STRING "Path to a custom linker program")
option(ENABLE_UBSAN "Enable the Clang Undefined Behavior Sanitizer" FALSE)
option(ENABLE_ASAN "Enable the Clang Address Sanitizer" FALSE)
option(ENABLE_TSAN "Enable the Clang Thread Sanitizer" FALSE)
option(ENABLE_COVERAGE "Enable Code Coverage" FALSE)
option(ENABLE_ASSERTIONS "Enable debug assertions" FALSE)
option(ENABLE_GDB_SYMBOLS "Enable GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
option(ENABLE_STRICT_WERROR "Enable stricter -Werror" FALSE)
option(ENABLE_WERROR "Enable -Werror" FALSE)
option(ENABLE_FEX_ALLOCATOR "Enable allocator for FEX" TRUE)
option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enable jemalloc glibc allocator" TRUE)
option(ENABLE_OFFLINE_TELEMETRY "Enable FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enable time trace compile option" FALSE)
option(ENABLE_LIBCXX "Use LLVM's libc++ instead of the GNU libstdc++" FALSE)
option(ENABLE_CCACHE "Enable ccache for build caching" TRUE)
option(ENABLE_VIXL_SIMULATOR "Use the VIXL simulator for emulation (only useful for CI testing)" FALSE)
option(ENABLE_VIXL_DISASSEMBLER "Enable debug disassembler output with VIXL" FALSE)
option(ENABLE_ZYDIS "Enable x86/x86-64 guest disassembler output with Zydis" FALSE)
option(USE_LEGACY_BINFMTMISC "Use legacy method of setting up binfmt_misc" FALSE)
option(ENABLE_FEXCORE_PROFILER "Enable FEXCore's timeline profiling capabilities" FALSE)
set(FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for FEXCore's profiler")
set_property(CACHE FEXCORE_PROFILER_BACKEND PROPERTY STRINGS gpuvis tracy)
set(USE_LINKER "" CACHE STRING "Allow overriding the linker path directly")
option(ENABLE_UBSAN "Enables Clang UBSAN" FALSE)
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
option(ENABLE_COVERAGE "Enables Coverage" FALSE)
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
option(ENABLE_WERROR "Enables -Werror" FALSE)
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enables jemalloc glibc allocator" TRUE)
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_VIXL_SIMULATOR "Enable use of VIXL simulator for emulation (only useful for CI testing)" FALSE)
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for the FEXCore profiler (gpuvis, tracy)")
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
option(USE_PDB_DEBUGINFO "Build debug info in PDB format" FALSE)
option(BUILD_STEAM_SUPPORT "Enable Steam integration" FALSE)
set(X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
set(X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
set(X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
set(DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
set(HOSTLIBS_DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
set (X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
set (DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
if (NOT DATA_DIRECTORY)
set(DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
endif()
include(GNUInstallDirs)
if (NOT HOSTLIBS_DATA_DIRECTORY)
set(HOSTLIBS_DATA_DIRECTORY "${CMAKE_INSTALL_FULL_LIBDIR}/fex-emu")
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
if (NOT CONTAINS_MINGW EQUAL -1)
message (STATUS "Mingw build")
set (MINGW_BUILD TRUE)
set (ENABLE_JEMALLOC TRUE)
set (ENABLE_JEMALLOC_GLIBC_ALLOC FALSE)
endif()
## Platform Checks ##
# Only 64-bit Linux and Windows are supported
# NB: SIZEOF_VOID_P is in bytes, not bits
# On 32-bit systems this is set to 4
if (NOT CMAKE_SIZEOF_VOID_P EQUAL 8)
message(FATAL_ERROR "Unsupported pointer size ${CMAKE_SIZEOF_VOID_P}."
" FEX only supports 64-bit (8-byte pointer) systems."
" If you believe this is in error, file an issue.")
elseif (NOT (WIN32 OR CMAKE_SYSTEM_NAME STREQUAL "Linux"))
message(FATAL_ERROR "Unsupported system type ${CMAKE_SYSTEM_NAME}."
" FEX only supports Linux and Windows."
" If you believe this is in error, file an issue.")
endif()
## Compiler Checks ##
# GCC and MSVC are unsupported
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
message(FATAL_ERROR "FEX doesn't support GCC! Use Clang instead.")
elseif (MSVC)
message(FATAL_ERROR "FEX doesn't support MSVC! Use Clang on MinGW instead.")
elseif (MINGW)
message(STATUS "Building for MinGW")
set(ENABLE_FEX_ALLOCATOR TRUE)
set(ENABLE_JEMALLOC_GLIBC_ALLOC FALSE)
else ()
message(STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
set(CLANG_MINIMUM_VERSION 13.0)
if (NOT MINGW_BUILD)
message (STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
set (CLANG_MINIMUM_VERSION 13.0)
if (CMAKE_CXX_COMPILER_VERSION VERSION_LESS ${CLANG_MINIMUM_VERSION})
message(FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
message (FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
endif()
endif()
## Architecture Handling ##
string(TOLOWER ${CMAKE_SYSTEM_PROCESSOR} processor)
if (processor MATCHES "x86|amd64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
message(FATAL_ERROR
" FEX doesn't support compiling for x86-64 hosts!"
" This is /only/ a supported configuration for FEX CI and nothing else!")
else()
message(STATUS "x86_64 debug build")
endif()
set(ARCHITECTURE_x86_64 1)
add_compile_definitions(ARCHITECTURE_x86_64=1)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
elseif (processor MATCHES "^aarch64|^arm64|^armv8\.*")
set(ARCHITECTURE_arm64 1)
add_compile_definitions(ARCHITECTURE_arm64=1)
# arm64ec needs to define both arm64 and arm64ec
if (processor MATCHES "^arm64ec")
set(ARCHITECTURE_arm64ec 1)
add_compile_definitions(ARCHITECTURE_arm64ec=1)
endif()
endif()
if (NOT (ARCHITECTURE_arm64 OR ARCHITECTURE_arm64ec OR ARCHITECTURE_x86_64))
message(FATAL_ERROR "Unsupported processor type ${processor}."
" If you believe this is in error, file an issue.")
endif()
if (BUILD_STEAM_SUPPORT)
add_compile_definitions(FEX_STEAM_SUPPORT=1)
endif()
if (ENABLE_FEXCORE_PROFILER)
add_compile_definitions(ENABLE_FEXCORE_PROFILER=1)
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
add_compile_definitions(FEXCORE_PROFILER_BACKEND=1)
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
elseif (FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
add_compile_definitions(FEXCORE_PROFILER_BACKEND=2)
add_compile_definitions(TRACY_ENABLE=1)
add_definitions(-DFEXCORE_PROFILER_BACKEND=2)
add_definitions(-DTRACY_ENABLE=1)
# Required so that Tracy will only start in the selected guest application
add_compile_definitions(TRACY_MANUAL_LIFETIME=1)
add_compile_definitions(TRACY_DELAYED_INIT=1)
add_definitions(-DTRACY_MANUAL_LIFETIME=1)
add_definitions(-DTRACY_DELAYED_INIT=1)
# This interferes with FEX's signal handling
add_compile_definitions(TRACY_NO_CRASH_HANDLER=1)
add_definitions(-DTRACY_NO_CRASH_HANDLER=1)
# Tracy can gather call stack samples in regular intervals, but this
# isn't useful for us since it would usually sample opaque JIT code
add_compile_definitions(TRACY_NO_SAMPLING=1)
add_definitions(-DTRACY_NO_SAMPLING=1)
# This pulls in libbacktrace which allocators in global constructors (before FEX can set up its allocator hooks)
add_compile_definitions(TRACY_NO_CALLSTACK=1)
if (MINGW)
message(FATAL_ERROR "Tracy profiler not supported on MinGW")
add_definitions(-DTRACY_NO_CALLSTACK=1)
if (MINGW_BUILD)
message(FATAL_ERROR "Tracy profiler not supported")
endif()
else()
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
@@ -152,7 +91,7 @@ if (ENABLE_JEMALLOC_GLIBC_ALLOC AND ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
endif()
if (ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
add_compile_definitions(GLIBC_ALLOCATOR_FAULT=1)
add_definitions(-DGLIBC_ALLOCATOR_FAULT=1)
endif()
# uninstall target
@@ -167,10 +106,9 @@ if(NOT TARGET uninstall)
endif()
# These options are meant for package management
set(TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
set(TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
set(OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version")
set(OVERRIDE_HASH "detect" CACHE STRING "Override the FEX git hash")
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
set (OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version in the format of <MMYY>{.<REV>}")
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
if (CMAKE_BUILD_TYPE MATCHES "DEBUG")
@@ -179,14 +117,15 @@ endif()
if (ENABLE_ASSERTIONS)
message(STATUS "Assertions enabled")
add_compile_definitions(ASSERTIONS_ENABLED=1)
add_definitions(-DASSERTIONS_ENABLED=1)
endif()
if (ENABLE_GDB_SYMBOLS)
message(STATUS "GDBSymbols support enabled")
add_compile_definitions(GDB_SYMBOLS_ENABLED=1)
add_definitions(-DGDB_SYMBOLS_ENABLED=1)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
@@ -196,7 +135,33 @@ cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
include(CheckPIESupported)
check_pie_supported()
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION ${ENABLE_LTO})
if (ENABLE_LTO)
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION TRUE)
else()
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION FALSE)
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
message(FATAL_ERROR
" FEX-Emu doesn't support compiling for x86-64 hosts!"
" This is /only/ a supported configuration for FEX CI and nothing else!")
endif()
set(_M_X86_64 1)
add_definitions(-D_M_X86_64=1)
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
set(_M_ARM_64 1)
add_definitions(-D_M_ARM_64=1)
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
set(_M_ARM_64EC 1)
add_definitions(-D_M_ARM_64EC=1)
endif()
include(CheckCXXSourceCompiles)
set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
@@ -212,33 +177,23 @@ check_cxx_source_compiles(
HAS_CLANG_PRESERVE_ALL)
unset(CMAKE_REQUIRED_FLAGS)
if (HAS_CLANG_PRESERVE_ALL)
if (MINGW)
if (MINGW_BUILD)
message(STATUS "Ignoring broken clang::preserve_all support")
set(HAS_CLANG_PRESERVE_ALL FALSE)
else()
message(STATUS "Has clang::preserve_all")
endif()
endif()
endif ()
if (ARCHITECTURE_arm64 AND HAS_CLANG_PRESERVE_ALL)
add_compile_definitions("FEX_PRESERVE_ALL_ATTR=__attribute__((preserve_all))" "FEX_HAS_PRESERVE_ALL_ATTR=1")
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
add_definitions("-DFEX_PRESERVE_ALL_ATTR=__attribute__((preserve_all))" "-DFEX_HAS_PRESERVE_ALL_ATTR=1")
else()
add_compile_definitions("FEX_PRESERVE_ALL_ATTR=" "FEX_HAS_PRESERVE_ALL_ATTR=0")
add_definitions("-DFEX_PRESERVE_ALL_ATTR=" "-DFEX_HAS_PRESERVE_ALL_ATTR=0")
endif()
check_cxx_source_compiles(
"
#define _GNU_SOURCE
#include <errno.h>
int main() {
return program_invocation_name == nullptr;
}"
HAS_PROGRAM_INVOCATION_NAME)
add_compile_definitions("HAS_PROGRAM_INVOCATION_NAME=${HAS_PROGRAM_INVOCATION_NAME}")
if (ENABLE_VIXL_SIMULATOR)
# We can run the simulator on both x86-64 or AArch64 hosts
add_compile_definitions(VIXL_SIMULATOR=1 VIXL_INCLUDE_SIMULATOR_AARCH64=1)
add_definitions(-DVIXL_SIMULATOR=1 -DVIXL_INCLUDE_SIMULATOR_AARCH64=1)
endif()
if (ENABLE_CCACHE)
@@ -259,7 +214,7 @@ if (ENABLE_COMPILE_TIME_TRACE)
link_libraries(-ftime-trace)
endif()
set(PTHREAD_LIB pthread)
set (PTHREAD_LIB pthread)
if (USE_LINKER)
message(STATUS "Overriding linker to: ${USE_LINKER}")
@@ -274,7 +229,7 @@ endif()
if (NOT ENABLE_OFFLINE_TELEMETRY)
# Disable FEX offline telemetry entirely if asked
add_compile_definitions(FEX_DISABLE_TELEMETRY=1)
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
endif()
if (ENABLE_UBSAN)
@@ -285,13 +240,13 @@ if (ENABLE_UBSAN)
# that are regularly access unaligned.
# function: syscalls cast function pointers to void (*)(unsigned long...), causing warnings
# related to this access.
add_compile_definitions(ENABLE_UBSAN=1)
add_definitions(-DENABLE_UBSAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=undefined -fno-sanitize=alignment -fno-sanitize=function -fno-sanitize-recover=undefined)
link_libraries(-fno-omit-frame-pointer -fsanitize=undefined -fno-sanitize=alignment -fno-sanitize=function -fno-sanitize-recover=undefined)
endif()
if (ENABLE_ASAN)
add_compile_definitions(ENABLE_ASAN=1)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
link_libraries(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
endif()
@@ -311,20 +266,20 @@ if (ENABLE_JEMALLOC_GLIBC_ALLOC)
# Required for thunks to work.
# All host native libraries will use this allocator, while *most* other FEX internal allocations will use the other jemalloc allocator.
add_subdirectory(External/jemalloc_glibc/)
elseif (NOT MINGW)
message(STATUS
elseif (NOT MINGW_BUILD)
message (STATUS
" jemalloc glibc allocator disabled!\n"
" This is not a recommended configuration!\n"
" This will very explicitly break thunk execution!\n"
" Use at your own risk!")
endif()
if (ENABLE_FEX_ALLOCATOR)
# The rpmalloc subproject that all FEXCore fextl objects allocate through.
add_subdirectory(External/rpmalloc/)
elseif (NOT MINGW)
if (ENABLE_JEMALLOC)
# The jemalloc subproject that all FEXCore fextl objects allocate through.
add_subdirectory(External/jemalloc/)
elseif (NOT MINGW_BUILD)
message (STATUS
" FEX allocator is disabled!\n"
" jemalloc disabled!\n"
" This is not a recommended configuration!\n"
" This will very explicitly break 32-bit application execution!\n"
" Use at your own risk!")
@@ -335,66 +290,46 @@ if (USE_PDB_DEBUGINFO)
add_link_options(-g -Wl,--pdb=)
endif()
set(CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set(CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set(CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set(CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
## Modules ##
list(APPEND CMAKE_MODULE_PATH ${CMAKE_SOURCE_DIR}/Data/CMake/)
include_directories(External/robin-map/include/)
include(LinkerGC)
## Externals ##
find_package(unordered_dense QUIET CONFIG)
if (NOT unordered_dense_FOUND)
add_subdirectory(External/unordered_dense)
endif()
include(CTest)
if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
add_subdirectory(External/vixl/)
endif()
if (ENABLE_ZYDIS)
find_package(Zycore 1.5 MODULE QUIET)
find_package(Zydis 4.0 MODULE QUIET)
if (TARGET Zydis::Zydis AND TARGET Zycore::Zycore)
message(STATUS "Using system Zydis")
else()
set(ZYDIS_BUILD_TOOLS OFF CACHE BOOL "" FORCE)
set(ZYDIS_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE)
message(STATUS "Using bundled Zydis")
add_subdirectory(External/zydis/)
endif()
include_directories(SYSTEM External/vixl/src/)
endif()
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
add_subdirectory(External/tracy)
endif()
find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
# This means we were attempted to get compiled with GCC
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
set(BUILD_SHARED_LIBS OFF)
if (NOT CMAKE_CROSSCOMPILING)
find_package(xxhash MODULE QUIET)
endif()
if (NOT TARGET xxHash::xxhash)
pkg_search_module(xxhash IMPORTED_TARGET xxhash libxxhash)
if (TARGET PkgConfig::xxhash AND NOT CMAKE_CROSSCOMPILING)
add_library(xxHash::xxhash ALIAS PkgConfig::xxhash)
else()
set(XXHASH_BUNDLED_MODE TRUE)
set(XXHASH_BUILD_XXHSUM FALSE)
add_subdirectory(External/xxhash/cmake_unofficial/)
endif()
add_compile_options(-Wno-trigraphs)
add_compile_definitions(GLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
if (BUILD_TESTING)
if (BUILD_TESTS)
find_package(Catch2 3 QUIET)
if (NOT Catch2_FOUND)
add_subdirectory(External/Catch2/)
@@ -404,9 +339,6 @@ if (BUILD_TESTING)
endif()
include(Catch)
else ()
# Override any previously generated test list to avoid running stale test binaries
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
endif()
find_package(fmt QUIET)
@@ -416,13 +348,8 @@ if (NOT fmt_FOUND)
add_subdirectory(External/fmt/)
endif()
find_package(range-v3 QUIET)
if (NOT range-v3_FOUND)
add_subdirectory(External/range-v3/)
target_compile_definitions(range-v3 INTERFACE RANGES_DISABLE_DEPRECATED_WARNINGS)
endif()
add_subdirectory(External/tiny-json/)
include_directories(External/tiny-json/)
include_directories(Source/)
include_directories("${CMAKE_BINARY_DIR}/Source/")
@@ -465,7 +392,7 @@ if (NOT TUNE_ARCH STREQUAL "generic")
endif()
if (TUNE_CPU STREQUAL "native")
if(ARCHITECTURE_arm64)
if(_M_ARM_64)
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
@@ -479,13 +406,6 @@ if (TUNE_CPU STREQUAL "native")
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/NeedDisabledSVE.py"
RESULT_VARIABLE NEEDS_SVE_DISABLED)
if (NEEDS_SVE_DISABLED)
message(STATUS "Platform has bugged SVE. Disabling")
set(AARCH64_CPU "cortex-a78")
endif()
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${AARCH64_CPU}")
@@ -506,54 +426,8 @@ elseif (NOT TUNE_CPU STREQUAL "none")
endif()
endif()
set(GIT_DESCRIBE_STRING "FEX-Unknown")
if (OVERRIDE_VERSION STREQUAL "detect")
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE)
endif()
else()
set(GIT_DESCRIBE_STRING "${OVERRIDE_VERSION}")
endif()
set(GIT_HASH "Unknown")
if (OVERRIDE_HASH STREQUAL "detect")
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} rev-parse HEAD
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_HASH
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE)
endif()
else()
set(GIT_HASH "${OVERRIDE_HASH}")
endif()
message(STATUS "FEX version: ${GIT_DESCRIBE_STRING}")
message(STATUS "FEX commit: ${GIT_HASH}")
# Prepends 0x to every two-character sequence in the hash,
# OR the final character of the hash, to plumb it for C++ usage. e.g.:
# -DOVERRIDE_HASH=123456aa => 0x12, 0x34, 0x56, 0xaa,
# -DOVERRIDE_HASH=12345678a => 0x12, 0x34, 0x56, 0x78, 0xa,
string(REGEX
REPLACE "(..|.$)" "0x\\1, "
GIT_HASH_ARRAY "${GIT_HASH}")
if (ENABLE_IWYU)
find_program(IWYU_EXE
NAMES iwyu include-what-you-use)
find_program(IWYU_EXE "iwyu")
if (IWYU_EXE)
message(STATUS "IWYU enabled")
set(CMAKE_CXX_INCLUDE_WHAT_YOU_USE "${IWYU_EXE}")
@@ -562,10 +436,15 @@ endif()
add_compile_options(-Wall)
if (BUILD_TESTING)
include(CTest)
if (BUILD_TESTS)
message(STATUS "Unit tests are enabled")
if (NOT BUILD_TESTING)
# CMake checks this variable before generating CTestTestfile.cmake
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
endif()
set(TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
if (TEST_JOB_COUNT)
message(STATUS "Running tests with ${TEST_JOB_COUNT} jobs")
elseif(CMAKE_VERSION VERSION_LESS "3.29")
@@ -580,16 +459,13 @@ add_subdirectory(FEXHeaderUtils/)
add_subdirectory(CodeEmitter/)
add_subdirectory(FEXCore/)
if (ARCHITECTURE_arm64 AND NOT MINGW AND NOT BUILD_STEAM_SUPPORT)
if (_M_ARM_64 AND NOT MINGW_BUILD)
# Binfmt_misc files must be installed prior to Source/ installs
add_subdirectory(Data/binfmts/)
endif()
add_subdirectory(Source/)
if (NOT BUILD_STEAM_SUPPORT)
add_subdirectory(Data/AppConfig/)
endif()
add_subdirectory(Data/AppConfig/)
# Install the ThunksDB file
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
@@ -597,16 +473,15 @@ file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.js
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/)
endforeach()
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
if (BUILD_THUNKS)
set(FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
set (FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
add_subdirectory(ThunkLibs/Generator)
# Thunk targets for both host libraries and IDE integration
@@ -633,7 +508,8 @@ if (BUILD_THUNKS)
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
DEPENDS thunkgen)
DEPENDS thunkgen
)
ExternalProject_Add(guest-libs-32
PREFIX guest-libs-32
@@ -651,36 +527,106 @@ if (BUILD_THUNKS)
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
DEPENDS thunkgen)
DEPENDS thunkgen
)
install(
CODE "message(\"-- Installing: guest-libs\")"
CODE "MESSAGE(\"-- Installing: guest-libs\")"
CODE "
execute_process(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest)"
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)"
DEPENDS guest-libs
COMPONENT Runtime)
)
install(
CODE "message(\"-- Installing: guest-libs-32\")"
CODE "MESSAGE(\"-- Installing: guest-libs-32\")"
CODE "
execute_process(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32)"
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)"
DEPENDS guest-libs-32
COMPONENT Runtime)
)
add_custom_target(uninstall_guest-libs
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)
add_custom_target(uninstall_guest-libs-32
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)
add_dependencies(uninstall uninstall_guest-libs)
add_dependencies(uninstall uninstall_guest-libs-32)
endif()
if (BUILD_STEAM_SUPPORT)
add_subdirectory(Source/Steam/)
set(FEX_VERSION_MAJOR "0")
set(FEX_VERSION_MINOR "0")
set(FEX_VERSION_PATCH "0")
if (OVERRIDE_VERSION STREQUAL "detect")
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
RESULT_VARIABLE GIT_ERROR
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
if (NOT ${GIT_ERROR} EQUAL 0)
# Likely built in a way that doesn't have tags
# Setup a version tag that is unknown
set(GIT_DESCRIBE_STRING "FEX-0000")
endif()
endif()
else()
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
endif()
# Parse the version here
# Change something like `FEX-2106.1-76-<hash>` in to a list
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
# Extract the `2106.1` element
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
# Change `2106.1` in to a list
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
# Calculate list size
list(LENGTH DESCRIBE_LIST LIST_SIZE)
# Pull out the major version
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
# Minor version only exists if there is a .1 at the end
# eg: 2106 versus 2106.1
if (LIST_SIZE GREATER 1)
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
endif()
# Package creation
set (CPACK_GENERATOR "DEB")
set (CPACK_PACKAGE_NAME fex-emu)
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/Description.txt")
# Debian defines
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/triggers")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
# binfmt_misc conflicts with qemu-user-static
# We also only install binfmt_misc on aarch64 hosts
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
endif()
include (CPack)
+1 -1
View File
@@ -129,4 +129,4 @@
"variables": []
}
]
}
}
+43 -69
View File
@@ -36,31 +36,24 @@ public:
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
void adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (IsADRRange(Imm)) {
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
return BranchEncodeSucceeded::Success;
}
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
// Can't encode.
return BranchEncodeSucceeded::Failure;
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, ForwardLabel* Label) {
void adr(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADR});
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return adr(rd, &Label->Backward);
adr(rd, &Label->Backward);
} else {
return adr(rd, &Label->Forward);
adr(rd, &Label->Forward);
}
}
@@ -69,53 +62,38 @@ public:
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
void adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) {
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
void adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADRP});
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return adrp(rd, &Label->Backward);
adrp(rd, &Label->Backward);
} else {
return adrp(rd, &Label->Forward);
adrp(rd, &Label->Forward);
}
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
const auto SLocation = reinterpret_cast<int64_t>(Label->Location);
const auto ULocation = std::bit_cast<uint64_t>(SLocation);
const int64_t Imm = SLocation - (GetCursorAddress<int64_t>());
const auto UImm = std::bit_cast<uint64_t>(Imm);
void LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
if (IsADRRange(Imm)) {
// If the range is in ADR range then we can just use ADR.
return adr(rd, Label);
}
if (IsADRPRange(Imm)) {
const int64_t ADRPImm = (SLocation & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
adr(rd, Label);
} else if (IsADRPRange(Imm)) {
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
// If the range is in the ADRP range then we can use ADRP.
const bool NeedsOffset = !IsADRPAligned(ULocation);
const uint64_t AlignedOffset = ULocation & 0xFFFULL;
bool NeedsOffset = !IsADRPAligned(reinterpret_cast<uint64_t>(Label->Location));
uint64_t AlignedOffset = reinterpret_cast<uint64_t>(Label->Location) & 0xFFFULL;
// First emit ADRP
adrp(rd, ADRPImm >> 12);
@@ -124,33 +102,23 @@ public:
// Now even an add
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
}
return BranchEncodeSucceeded::Success;
} else {
LOGMAN_MSG_A_FMT("Unscaled offset too large");
FEX_UNREACHABLE;
}
// Stinky path, we need to load the address as a sequence of movz+movk+movk
movz(ARMEmitter::Size::i64Bit, rd, (UImm >> 32) & 0xFFFF, 32);
movk(ARMEmitter::Size::i64Bit, rd, (UImm >> 16) & 0xFFFF, 16);
movk(ARMEmitter::Size::i64Bit, rd, UImm & 0xFFFF);
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
// Emit a register index and two nops. These will be backpatched.
// Emit a register index and a nop. These will be backpatched.
dc32(rd.Idx());
nop();
nop();
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return LongAddressGen(rd, &Label->Backward);
LongAddressGen(rd, &Label->Backward);
} else {
return LongAddressGen(rd, &Label->Forward);
LongAddressGen(rd, &Label->Forward);
}
}
@@ -206,7 +174,7 @@ public:
// Logical immediate
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
and_(s, rd, rn, n, immr, imms);
}
@@ -217,7 +185,7 @@ public:
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
ands(s, rd, rn, n, immr, imms);
}
@@ -228,14 +196,14 @@ public:
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
orr(s, rd, rn, n, immr, imms);
}
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
eor(s, rd, rn, n, immr, imms);
}
@@ -365,7 +333,7 @@ public:
bfi(s, rd, Reg::zr, lsb, width);
}
void bfxil(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
const auto reg_size_bits = RegSizeInBits(s);
[[maybe_unused]] const auto reg_size_bits = RegSizeInBits(s);
const auto lsb_p_width = lsb + width;
LOGMAN_THROW_A_FMT(width >= 1, "bfxil needs width >= 1");
@@ -894,6 +862,12 @@ public:
}
private:
static constexpr Condition InvertCondition(Condition cond) {
// These behave as always, so it makes no sense to allow inverting these.
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
}
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b001'0010'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
@@ -1003,7 +977,7 @@ private:
}
void xbfiz_helper(bool is_signed, ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
const auto lsb_p_width = lsb + width;
[[maybe_unused]] const auto lsb_p_width = lsb + width;
const auto reg_size_bits = RegSizeInBits(s);
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}", lsb_p_width, reg_size_bits, lsb, width);
File diff suppressed because it is too large. Load diff
+64 -123
View File
@@ -20,31 +20,23 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm);
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
void b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
void b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
void b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return b(Cond, &Label->Backward);
b(Cond, &Label->Backward);
} else {
return b(Cond, &Label->Forward);
b(Cond, &Label->Forward);
}
}
@@ -53,32 +45,24 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm);
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
void bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
void bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return bc(Cond, &Label->Backward);
bc(Cond, &Label->Backward);
} else {
return bc(Cond, &Label->Forward);
bc(Cond, &Label->Forward);
}
}
@@ -114,32 +98,25 @@ public:
UnconditionalBranch(Op, Imm);
}
[[nodiscard]] BranchEncodeSucceeded b(const BackwardLabel* Label) {
void b(const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0001'01 << 26;
// Can't encode.
return BranchEncodeSucceeded::Failure;
UnconditionalBranch(Op, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded b(ForwardLabel* Label) {
void b(ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded b(BiDirectionalLabel* Label) {
void b(BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return b(&Label->Backward);
b(&Label->Backward);
} else {
return b(&Label->Forward);
b(&Label->Forward);
}
}
@@ -149,33 +126,25 @@ public:
UnconditionalBranch(Op, Imm);
}
[[nodiscard]] BranchEncodeSucceeded bl(const BackwardLabel* Label) {
void bl(const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b1001'01 << 26;
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
UnconditionalBranch(Op, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded bl(ForwardLabel* Label) {
void bl(ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded bl(BiDirectionalLabel* Label) {
void bl(BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return bl(&Label->Backward);
bl(&Label->Backward);
} else {
return bl(&Label->Forward);
bl(&Label->Forward);
}
}
@@ -186,35 +155,28 @@ public:
CompareAndBranch(Op, s, rt, Imm);
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0100 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return cbz(s, rt, &Label->Backward);
cbz(s, rt, &Label->Backward);
} else {
return cbz(s, rt, &Label->Forward);
cbz(s, rt, &Label->Forward);
}
}
@@ -224,35 +186,28 @@ public:
CompareAndBranch(Op, s, rt, Imm);
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0101 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return cbnz(s, rt, &Label->Backward);
cbnz(s, rt, &Label->Backward);
} else {
return cbnz(s, rt, &Label->Forward);
cbnz(s, rt, &Label->Forward);
}
}
@@ -262,35 +217,28 @@ public:
TestAndBranch(Op, rt, Bit, Imm);
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0110 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return tbz(rt, Bit, &Label->Backward);
tbz(rt, Bit, &Label->Backward);
} else {
return tbz(rt, Bit, &Label->Forward);
tbz(rt, Bit, &Label->Forward);
}
}
@@ -299,34 +247,27 @@ public:
TestAndBranch(Op, rt, Bit, Imm);
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0111 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return tbnz(rt, Bit, &Label->Backward);
tbnz(rt, Bit, &Label->Backward);
} else {
return tbnz(rt, Bit, &Label->Forward);
tbnz(rt, Bit, &Label->Forward);
}
}
-1
View File
@@ -53,7 +53,6 @@ public:
if (!CurrentAlignment) {
return;
}
std::memset(CurrentOffset, 0, Size - CurrentAlignment);
CurrentOffset += Size - CurrentAlignment;
}
+31 -87
View File
@@ -12,7 +12,6 @@
#include <CodeEmitter/Registers.h>
#include <array>
#include <bit>
#include <cstdint>
#include <utility>
#include <type_traits>
@@ -87,14 +86,6 @@ constexpr size_t SubRegSizeInBits(SubRegSize size) {
return size_t {8} << FEXCore::ToUnderlying(size);
}
// Many floating point operations constrain their element sizes to the
// main three float sizes half, single, and double precision. This just
// combines all the checks together for brevity.
[[nodiscard]]
constexpr bool IsStandardFloatSize(SubRegSize size) {
return size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit;
}
/* This `ScalarRegSize` enum is used for most scalar float
* operations.
*
@@ -586,15 +577,6 @@ concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegi
template<typename T>
concept IsQOrDRegister = std::is_same_v<T, QRegister> || std::is_same_v<T, DRegister>;
template<typename T>
concept IsLabel = std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>;
enum class BranchEncodeSucceeded {
Success,
Failure,
};
// Whether or not a given set of vector registers are sequential
// in increasing order as far as the register file is concerned (modulo its size)
//
@@ -647,25 +629,19 @@ public:
// Bind a backward label to an address.
// Address that is bound is the current emitter location.
[[nodiscard]] bool Bind(BackwardLabel* Label) {
void Bind(BackwardLabel* Label) {
LOGMAN_THROW_A_FMT(Label->Location == nullptr, "Trying to bind a label twice");
Label->Location = GetCursorAddress<uint8_t*>();
// Always binds because it is only storing a location.
return true;
}
[[nodiscard]] bool Bind(const ForwardLabel::Reference* Label) {
void Bind(const ForwardLabel::Reference* Label) {
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
// Patch up the instructions
switch (Label->Type) {
case ForwardLabel::InstType::ADR: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!IsADRRange(Imm)) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
@@ -677,12 +653,7 @@ public:
case ForwardLabel::InstType::ADRP: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
@@ -692,13 +663,11 @@ public:
*Instruction = Inst;
break;
}
case ForwardLabel::InstType::B: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -708,13 +677,11 @@ public:
break;
}
case ForwardLabel::InstType::TEST_BRANCH: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -728,10 +695,7 @@ public:
case ForwardLabel::InstType::RELATIVE_LOAD: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -741,44 +705,38 @@ public:
break;
}
case ForwardLabel::InstType::LONG_ADDRESS_GEN: {
const auto* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
const auto ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
const auto ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
const auto ImmInstThree = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[2]);
const auto OriginalOffset = GetCursorOffset();
uint32_t* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
auto OriginalOffset = GetCursorOffset();
const auto InstOffset = GetCursorOffsetFromAddress(Instructions);
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
SetCursorOffset(InstOffset);
// We encoded the destination register in to the first instruction space.
// Read it back.
ARMEmitter::Register DestReg(Instructions[0]);
if (IsADRRange(ImmInstThree)) {
// If within ADR range from the third instruction, then we can emit NOP+NOP+ADR
if (IsADRRange(ImmInstTwo)) {
// If within ADR range from the second instruction, then we can emit NOP+ADR
nop();
nop();
adr(DestReg, static_cast<uint32_t>(ImmInstThree) & 0x7FFF);
} else if (IsADRPRange(ImmInstTwo)) {
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
} else if (IsADRPRange(ImmInstOne)) {
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
// First check if we are in non-offset range for second instruction.
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
// We can emit nop + nop + adrp
nop();
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstThree >> 12) & 0x7FFF);
} else {
// Not aligned, need nop + adrp + add
// We can emit nop + adrp
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstTwo & 0xFFF);
} else {
// Not aligned, need adrp + add
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
}
} else {
// Stinky path, we need to emit a movz+movk+movk sequence.
movz(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 32) & 0x7FFF, 32);
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 16) & 0xFFFF, 16);
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne) & 0xFFFF);
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
FEX_UNREACHABLE;
}
SetCursorOffset(OriginalOffset);
@@ -786,41 +744,27 @@ public:
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
return true;
}
// Bind a forward label to a location.
// This walks all the instructions in the label's vector.
// Then backpatching all instructions that have used the label.
[[nodiscard]] bool Bind(ForwardLabel* Label) {
bool Bound = true;
void Bind(ForwardLabel* Label) {
if (Label->FirstInst.Location) {
Bound &= Bind(&Label->FirstInst);
Bind(&Label->FirstInst);
}
for (auto& Inst : Label->Insts) {
Bound &= Bind(&Inst);
Bind(&Inst);
}
return Bound;
}
// Bind a bidirectional location to a location.
// Binds both forwards and backwards depending on how the label was used.
[[nodiscard]] bool Bind(BiDirectionalLabel* Label) {
bool Bound = true;
void Bind(BiDirectionalLabel* Label) {
if (!Label->Backward.Location) {
Bound &= Bind(&Label->Backward);
Bind(&Label->Backward);
}
Bound &= Bind(&Label->Forward);
return Bound;
}
static constexpr Condition InvertCondition(Condition cond) {
// These behave as always, so it makes no sense to allow inverting these.
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
Bind(&Label->Forward);
}
#include <CodeEmitter/VixlUtils.inl>
+51 -44
View File
@@ -60,7 +60,8 @@ public:
}
void fcmla(SubRegSize size, ZRegister zda, PRegisterMerge pv, ZRegister zn, ZRegister zm, Rotation rot) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pv <= PReg::p7.Merging(), "fcmla can only use p0 to p7");
uint32_t Op = 0b0110'0100'0000'0000'0000'0000'0000'0000;
@@ -75,7 +76,8 @@ public:
}
void fcadd(SubRegSize size, ZRegister zd, PRegisterMerge pv, ZRegister zn, ZRegister zm, Rotation rot) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pv <= PReg::p7.Merging(), "fcadd can only use p0 to p7");
LOGMAN_THROW_A_FMT(rot == Rotation::ROTATE_90 || rot == Rotation::ROTATE_270, "fcadd rotation may only be 90 or 270 degrees");
LOGMAN_THROW_A_FMT(zd == zn, "fcadd zd and zn must be the same register");
@@ -813,12 +815,16 @@ public:
// SVE Integer Misc - Unpredicated
// SVE floating-point trig select coefficient
void ftssel(SubRegSize size, ZRegister zd, ZRegister zn, ZRegister zm) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "ftssel may only use 16/32/64-bit element sizes");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "ftssel may only have "
"16-bit, 32-bit, or 64-bit "
"element sizes");
SVEIntegerMiscUnpredicated(0b00, zm.Idx(), FEXCore::ToUnderlying(size), zd, zn);
}
// SVE floating-point exponential accelerator
void fexpa(SubRegSize size, ZRegister zd, ZRegister zn) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "fexpa may only use 16/32/64-bit element sizes");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "fexpa may only have "
"16-bit, 32-bit, or 64-bit "
"element sizes");
SVEIntegerMiscUnpredicated(0b10, 0b00000, FEXCore::ToUnderlying(size), zd, zn);
}
// SVE constructive prefix (unpredicated)
@@ -1497,9 +1503,9 @@ public:
}
// SVE broadcast floating-point immediate (unpredicated)
void fdup(SubRegSize size, ZRegister zd, float Value) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Unsupported fmov size");
void fdup(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, float Value) {
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i16Bit || size == ARMEmitter::SubRegSize::i32Bit || size == ARMEmitter::SubRegSize::i64Bit,
"Unsupported fmov size");
uint32_t Imm {};
if (size == SubRegSize::i16Bit) {
LOGMAN_MSG_A_FMT("Unsupported");
@@ -1512,7 +1518,7 @@ public:
SVEBroadcastFloatImmUnpredicated(0b00, 0, Imm, size, zd);
}
void fmov(SubRegSize size, ZRegister zd, float Value) {
void fmov(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, float Value) {
fdup(size, zd, Value);
}
@@ -1541,7 +1547,7 @@ public:
void sqincp(SubRegSize size, XRegister rdn, PRegister pm) {
SVEIncDecPredicateCountScalar(0, 1, 0b10, 0b00, size, rdn, pm);
}
void sqincp(SubRegSize size, XRegister rdn, PRegister pm, WRegister wn) {
void sqincp(SubRegSize size, XRegister rdn, PRegister pm, [[maybe_unused]] WRegister wn) {
LOGMAN_THROW_A_FMT(rdn.Idx() == wn.Idx(), "rdn and wn must be the same");
SVEIncDecPredicateCountScalar(0, 1, 0b00, 0b00, size, rdn, pm);
}
@@ -1554,7 +1560,7 @@ public:
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm) {
SVEIncDecPredicateCountScalar(0, 1, 0b10, 0b10, size, rdn, pm);
}
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm, WRegister wn) {
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm, [[maybe_unused]] WRegister wn) {
LOGMAN_THROW_A_FMT(rdn.Idx() == wn.Idx(), "rdn and wn must be the same");
SVEIncDecPredicateCountScalar(0, 1, 0b00, 0b10, size, rdn, pm);
}
@@ -3296,7 +3302,7 @@ private:
const auto log2_size_bytes = FEXCore::ilog2(size_bytes);
// We can index up to 512-bit registers with dup
const auto max_index = (64U >> log2_size_bytes) - 1;
[[maybe_unused]] const auto max_index = (64U >> log2_size_bytes) - 1;
LOGMAN_THROW_A_FMT(Index <= max_index, "dup index ({}) too large. Must be within [0, {}].", Index, max_index);
// imm2:tsz make up a 7 bit wide field, with each increasing element size
@@ -3326,7 +3332,7 @@ private:
uint32_t shift = 0;
if (!is_uint8_imm) {
const bool is_uint16_imm = (imm >> 16) == 0;
[[maybe_unused]] const bool is_uint16_imm = (imm >> 16) == 0;
LOGMAN_THROW_A_FMT(is_uint16_imm, "Immediate ({}) must be a 16-bit value within [256, 65280]", imm);
LOGMAN_THROW_A_FMT((imm % 256) == 0, "Immediate ({}) must be a multiple of 256", imm);
@@ -3395,8 +3401,8 @@ private:
}
void SVEBroadcastFloatImmPredicated(SubRegSize size, ZRegister zd, PRegister pg, float value) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Unsupported fcpy/fmov size");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Unsupported fcpy/fmov "
"size");
uint32_t imm {};
if (size == SubRegSize::i16Bit) {
LOGMAN_MSG_A_FMT("Unsupported");
@@ -3572,7 +3578,7 @@ private:
// SVE2 floating-point pairwise operations
void SVEFloatPairwiseArithmetic(uint32_t opc, SubRegSize size, PRegister pg, ZRegister zd, ZRegister zn, ZRegister zm) {
LOGMAN_THROW_A_FMT(zd == zn, "zd needs to equal zn");
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Invalid float size");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Invalid float size");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0100'0001'0000'1000'0000'0000'0000;
@@ -3586,7 +3592,7 @@ private:
// SVE floating-point arithmetic (unpredicated)
void SVEFloatArithmeticUnpredicated(uint32_t opc, SubRegSize size, ZRegister zm, ZRegister zn, ZRegister zd) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Invalid float size");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Invalid float size");
uint32_t Instr = 0b0110'0101'0000'0000'0000'0000'0000'0000;
Instr |= FEXCore::ToUnderlying(size) << 22;
@@ -3694,7 +3700,7 @@ private:
// SVE floating-point arithmetic (predicated)
void SVEFloatArithmeticPredicated(uint32_t opc, SubRegSize size, PRegister pg, ZRegister zd, ZRegister zn, ZRegister zm) {
LOGMAN_THROW_A_FMT(zd == zn, "zn needs to equal zd");
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Invalid float size");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Invalid float size");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0101'0000'0000'1000'0000'0000'0000;
@@ -3722,7 +3728,9 @@ private:
}
void SVEFPRecursiveReduction(uint32_t opc, SubRegSize size, VRegister vd, PRegister pg, ZRegister zn) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "FP reduction operation can only use 16/32/64-bit element sizes");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "FP reduction operation can "
"only use 16-bit, 32-bit, "
"or 64-bit element sizes");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "FP reduction operation can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0101'0000'0000'0010'0000'0000'0000;
@@ -4104,7 +4112,7 @@ private:
// 0b111 - I - Current
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Unsupported size in {}", __func__);
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Unsupported size in {}", __func__);
uint32_t Instr = 0b0110'0101'0000'0000'1010'0000'0000'0000;
Instr |= FEXCore::ToUnderlying(size) << 22;
@@ -4152,7 +4160,7 @@ private:
const auto& op_data = mem_op.MetaType.ScalarVectorType;
const bool is_scaled = op_data.scale != 0;
const auto msize_value = FEXCore::ToUnderlying(msize);
[[maybe_unused]] const auto msize_value = FEXCore::ToUnderlying(msize);
LOGMAN_THROW_A_FMT(op_data.scale == 0 || op_data.scale == msize_value, "scale may only be 0 or {}", msize_value);
@@ -4266,7 +4274,7 @@ private:
const auto msize_value = FEXCore::ToUnderlying(msize);
const auto msize_bytes = 1U << msize_value;
const auto imm_limit = (32U << msize_value) - msize_bytes;
[[maybe_unused]] const auto imm_limit = (32U << msize_value) - msize_bytes;
const auto imm = mem_op.MetaType.VectorImmType.Imm;
const auto imm_to_encode = imm >> msize_value;
@@ -4332,8 +4340,8 @@ private:
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_A_FMT((imm % num_regs) == 0, "Offset must be a multiple of {}", num_regs);
const auto min_offset = -8 * num_regs;
const auto max_offset = 7 * num_regs;
[[maybe_unused]] const auto min_offset = -8 * num_regs;
[[maybe_unused]] const auto max_offset = 7 * num_regs;
LOGMAN_THROW_A_FMT(imm >= min_offset && imm <= max_offset,
"Invalid load/store offset ({}). Offset must be a multiple of {} and be within [{}, {}]", imm, num_regs, min_offset,
max_offset);
@@ -4440,8 +4448,8 @@ private:
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
const auto esize = static_cast<int>(16 << ssz);
const auto max_imm = (esize << 3) - esize;
const auto min_imm = -(max_imm + esize);
[[maybe_unused]] const auto max_imm = (esize << 3) - esize;
[[maybe_unused]] const auto min_imm = -(max_imm + esize);
LOGMAN_THROW_A_FMT((imm % esize) == 0, "imm ({}) must be a multiple of {}", imm, esize);
LOGMAN_THROW_A_FMT(imm >= min_imm && imm <= max_imm, "imm ({}) must be within [{}, {}]", imm, min_imm, max_imm);
@@ -4485,7 +4493,7 @@ private:
const auto msize_value = FEXCore::ToUnderlying(msize);
const auto data_size_bytes = 1U << msize_value;
const auto max_imm = (64U << msize_value) - data_size_bytes;
[[maybe_unused]] const auto max_imm = (64U << msize_value) - data_size_bytes;
LOGMAN_THROW_A_FMT((imm % data_size_bytes) == 0 && imm <= max_imm, "imm must be a multiple of {} and be within [0, {}]",
data_size_bytes, max_imm);
@@ -4713,7 +4721,7 @@ private:
void SVEFloatUnary(uint32_t opc, SubRegSize size, PRegister pg, ZRegister zn, ZRegister zd) {
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Unsupported size in {}", __func__);
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Unsupported size in {}", __func__);
uint32_t Instr = 0b0110'0101'0000'1100'1010'0000'0000'0000;
Instr |= FEXCore::ToUnderlying(size) << 22;
@@ -4801,7 +4809,8 @@ private:
}
void SVEFPUnaryOpsUnpredicated(uint32_t opc, SubRegSize size, ZRegister zd, ZRegister zn) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
uint32_t Instr = 0b0110'0101'0000'1000'0011'0000'0000'0000;
Instr |= FEXCore::ToUnderlying(size) << 22;
@@ -4812,7 +4821,8 @@ private:
}
void SVEFPSerialReductionPredicated(uint32_t opc, SubRegSize size, VRegister vd, PRegister pg, VRegister vn, ZRegister zm) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_A_FMT(vd == vn, "vn must be the same as vd");
@@ -4826,7 +4836,8 @@ private:
}
void SVEFPCompareWithZero(uint32_t eqlt, uint32_t ne, SubRegSize size, PRegister pd, PRegister pg, ZRegister zn) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0101'0001'0000'0010'0000'0000'0000;
@@ -4841,7 +4852,8 @@ private:
void SVEFPMultiplyAdd(uint32_t opc, SubRegSize size, ZRegister zd, PRegister pg, ZRegister zn, ZRegister zm) {
// NOTE: opc also includes the op0 bit (bit 15) like op0:opc, since the fields are adjacent
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0101'0010'0000'0000'0000'0000'0000;
@@ -4855,13 +4867,14 @@ private:
}
void SVEFPMultiplyAddIndexed(uint32_t op, SubRegSize size, ZRegister zda, ZRegister zn, ZRegister zm, uint32_t index) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT((size <= SubRegSize::i32Bit && zm <= ZReg::z7) || (size == SubRegSize::i64Bit && zm <= ZReg::z15),
"16-bit and 32-bit indexed variants may only use Zm between z0-z7\n"
"64-bit variants may only use Zm between z0-z15");
const auto Underlying = FEXCore::ToUnderlying(size);
const uint32_t IndexMax = (16 / (1U << Underlying)) - 1;
[[maybe_unused]] const uint32_t IndexMax = (16 / (1U << Underlying)) - 1;
LOGMAN_THROW_A_FMT(index <= IndexMax, "Index must be within 0-{}", IndexMax);
// Can be bit 20 or 19 depending on whether or not the element size is 64-bit.
@@ -5117,15 +5130,14 @@ private:
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
using FloatToEquivalentUInt = std::conditional_t<std::is_same_v<T, float>, uint32_t, uint64_t>;
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
// Determines if a floating-point value is capable of being converted
// into an 8-bit immediate. See pseudocode definition of VFPExpandImm
// in ARM A-profile reference manual for a general overview of how this was derived.
template<typename T>
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
[[nodiscard]]
[[nodiscard, maybe_unused]]
static bool IsValidFPValueForImm8(T value) {
const uint64_t bits = std::bit_cast<FloatToEquivalentUInt<T>>(value);
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
static constexpr std::array mantissa_masks {
@@ -5163,15 +5175,12 @@ private:
return true;
}
#endif
protected:
static uint32_t FP32ToImm8(float value) {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
#endif
const auto bits = std::bit_cast<uint32_t>(value);
const auto bits = FEXCore::BitCast<uint32_t>(value);
const auto sign = (bits & 0x80000000) >> 24;
const auto expb2 = (bits & 0x20000000) >> 23;
const auto b5_to_0 = (bits >> 19) & 0x3F;
@@ -5180,11 +5189,9 @@ protected:
}
static uint32_t FP64ToImm8(double value) {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
#endif
const auto bits = std::bit_cast<uint64_t>(value);
const auto bits = FEXCore::BitCast<uint64_t>(value);
const auto sign = (bits & 0x80000000'00000000) >> 56;
const auto expb2 = (bits & 0x20000000'00000000) >> 55;
const auto b5_to_0 = (bits >> 48) & 0x3F;
@@ -5208,7 +5215,7 @@ private:
uint32_t shift = 0;
if (!is_int8_imm) {
const int32_t imm16_limit = 32768;
const bool is_int16_imm = -imm16_limit <= imm && imm < imm16_limit;
[[maybe_unused]] const bool is_int16_imm = -imm16_limit <= imm && imm < imm16_limit;
LOGMAN_THROW_A_FMT(is_int16_imm, "Immediate ({}) must be a 16-bit value within [-32768, 32512]", imm);
LOGMAN_THROW_A_FMT((imm % 256) == 0, "Immediate ({}) must be a multiple of 256", imm);
+9 -7
View File
@@ -27,19 +27,21 @@ struct EmitterOps : Emitter {
public:
// Advanced SIMD scalar copy
void dup(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Index) {
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'01 << 10;
const uint32_t SizeImm = FEXCore::ToUnderlying(size);
const uint32_t IndexShift = SizeImm + 1;
const uint32_t ElementSize = 1U << SizeImm;
const uint32_t MaxIndex = 128U / (ElementSize * 8);
[[maybe_unused]] const uint32_t MaxIndex = 128U / (ElementSize * 8);
LOGMAN_THROW_A_FMT(Index < MaxIndex, "Index too large. Index={}, Max Index: {}", Index, MaxIndex);
const uint32_t imm5 = (Index << IndexShift) | ElementSize;
ASIMDScalarCopy(1, 1, imm5, 0b0000, rd, rn);
ASIMDScalarCopy(Op, 1, imm5, 0b0000, rd, rn);
}
void mov(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Index) {
void mov(ARMEmitter::ScalarRegSize size, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn, uint32_t Index) {
dup(size, rd, rn, Index);
}
@@ -1280,10 +1282,10 @@ public:
private:
// Advanced SIMD scalar copy
void ASIMDScalarCopy(uint32_t Q, uint32_t b28, uint32_t imm5, uint32_t imm4, VRegister rd, VRegister rn) {
uint32_t Instr = 0b0000'1110'0000'0000'0000'01U << 10;
void ASIMDScalarCopy(uint32_t Op, uint32_t Q, uint32_t imm5, uint32_t imm4, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn) {
uint32_t Instr = Op;
Instr |= Q << 30;
Instr |= b28 << 28;
Instr |= imm5 << 16;
Instr |= imm4 << 11;
Instr |= Encode_rn(rn);
@@ -1381,7 +1383,7 @@ private:
void ASIMDScalarXIndexedElement(uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rm, VRegister rn, VRegister rd, uint32_t index) {
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i8Bit, "Scalar size must not be 8-bit");
const auto invalid_bound = 16U >> FEXCore::ToUnderlying(size);
[[maybe_unused]] const auto invalid_bound = 16U >> FEXCore::ToUnderlying(size);
LOGMAN_THROW_A_FMT(index < invalid_bound, "Index ({}) must be within [0-{}]", index, invalid_bound - 1);
uint32_t Instr = 0b0101'1111'0000'0000'0000'0000'0000'0000;
+59 -11
View File
@@ -41,6 +41,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
[[maybe_unused]] constexpr auto kDRegSize = 64;
constexpr auto kWRegSize = 32;
constexpr auto kXRegSize = 64;
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) || (width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
@@ -128,8 +129,8 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
// Compute the repeat distance d, and set up a bitmask covering the basic
// unit of repetition (i.e. a word with the bottom d bits set). Also, in all
// of these cases the N bit of the output will be zero.
clz_a = std::countl_zero(a);
int clz_c = std::countl_zero(c);
clz_a = CountLeadingZeros(a, kXRegSize);
int clz_c = CountLeadingZeros(c, kXRegSize);
d = clz_a - clz_c;
mask = ((UINT64_C(1) << d) - 1);
out_n = 0;
@@ -150,7 +151,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
// of set bits in our word, meaning that we have the trivial case of
// d == 64 and only one 'repetition'. Set up all the same variables as in
// the general case above, and set the N bit in the output.
clz_a = std::countl_zero(a);
clz_a = CountLeadingZeros(a, kXRegSize);
d = 64;
mask = ~UINT64_C(0);
out_n = 1;
@@ -158,7 +159,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
}
// If the repeat period d is not a power of two, it can't be encoded.
if (!std::has_single_bit(uint32_t(d))) {
if (!IsPowerOf2(d)) {
return false;
}
@@ -178,7 +179,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
static const uint64_t multipliers[] = {
0x0000000000000001UL, 0x0000000100000001UL, 0x0001000100010001UL, 0x0101010101010101UL, 0x1111111111111111UL, 0x5555555555555555UL,
};
uint64_t multiplier = multipliers[std::countl_zero(uint64_t(d)) - 57];
uint64_t multiplier = multipliers[CountLeadingZeros(d, kXRegSize) - 57];
uint64_t candidate = (b - a) * multiplier;
if (value != candidate) {
@@ -193,7 +194,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
// Count the set bits in our basic stretch. The special case of clz(0) == -1
// makes the answer come out right for stretches that reach the very top of
// the word (e.g. numbers like 0xffffc00000000000).
int clz_b = (b == 0) ? -1 : std::countl_zero(b);
int clz_b = (b == 0) ? -1 : CountLeadingZeros(b, kXRegSize);
int s = clz_a - clz_b;
// Decide how many bits to rotate right by, to put the low bit of that basic
@@ -223,13 +224,9 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
// 11110s 2 UInt(s)
//
// So we 'or' (2 * -d) with our computed s to form imms.
if (n != nullptr) {
if ((n != NULL) || (imm_s != NULL) || (imm_r != NULL)) {
*n = out_n;
}
if (imm_s != nullptr) {
*imm_s = ((2 * -d) | (s - 1)) & 0x3f;
}
if (imm_r != nullptr) {
*imm_r = r;
}
@@ -284,6 +281,11 @@ INT_1_TO_63_LIST(DECLARE_IS_UINT_N)
private:
template<typename V>
static inline bool IsPowerOf2(V value) {
return (value != 0) && ((value & (value - 1)) == 0);
}
// Some compilers dislike negating unsigned integers,
// so we provide an equivalent.
template<typename T>
@@ -296,4 +298,50 @@ static inline uint64_t LowestSetBit(uint64_t value) {
return value & UnsignedNegate(value);
}
template<typename V>
static inline int CountLeadingZeros(V value, int width = (sizeof(V) * 8)) {
#if COMPILER_HAS_BUILTIN_CLZ
if (width == 32) {
return (value == 0) ? 32 : __builtin_clz(static_cast<unsigned>(value));
} else if (width == 64) {
return (value == 0) ? 64 : __builtin_clzll(value);
}
#endif
return CountLeadingZerosFallBack(value, width);
}
static inline int CountLeadingZerosFallBack(uint64_t value, int width) {
LOGMAN_THROW_A_FMT(IsPowerOf2(width) && (width <= 64), "Invalid width");
if (value == 0) {
return width;
}
int count = 0;
value = value << (64 - width);
if ((value & UINT64_C(0xffffffff00000000)) == 0) {
count += 32;
value = value << 32;
}
if ((value & UINT64_C(0xffff000000000000)) == 0) {
count += 16;
value = value << 16;
}
if ((value & UINT64_C(0xff00000000000000)) == 0) {
count += 8;
value = value << 8;
}
if ((value & UINT64_C(0xf000000000000000)) == 0) {
count += 4;
value = value << 4;
}
if ((value & UINT64_C(0xc000000000000000)) == 0) {
count += 2;
value = value << 2;
}
if ((value & UINT64_C(0x8000000000000000)) == 0) {
count += 1;
}
count += (value == 0);
return count;
}
public:
+7 -6
View File
@@ -4,8 +4,7 @@ file(GLOB GEN_CONFIG_SOURCES CONFIGURE_DEPENDS *.json.in)
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
# Any configuration file json file that needs to be generated
@@ -15,10 +14,12 @@ foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WLE)
# Configure it
configure_file(${GEN_CONFIG_SRC} ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
configure_file(
${GEN_CONFIG_SRC}
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
# Then install the configured json
install(FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
install(
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
+3
View File
@@ -0,0 +1,3 @@
x86 and x86-64 Linux emulator
FEX allows you to run x86 applications on ARM64 Linux devices. It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
+18
View File
@@ -0,0 +1,18 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Setup binfmt_misc
update-binfmts --import FEX-x86
update-binfmts --import FEX-x86_64
}
# Install FEXInterpreter hardlink
# Needs to be done before setting up binfmt_misc
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
+17
View File
@@ -0,0 +1,17 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Uninstall
update-binfmts --unimport FEX-x86
update-binfmts --unimport FEX-x86_64
}
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
# Remove FEXInterpreter hardlink
unlink /usr/bin/FEXInterpreter
+1
View File
@@ -0,0 +1 @@
activate-noawait ldconfig
-23
View File
@@ -1,23 +0,0 @@
# SPDX-License-Identifier: MIT
if (CMAKE_CROSSCOMPILING)
return()
endif()
include(FindPackageHandleStandardArgs)
find_package(Zycore QUIET CONFIG)
if (Zycore_CONSIDERED_CONFIGS)
find_package_handle_standard_args(Zycore CONFIG_MODE)
else()
find_package(PkgConfig QUIET)
pkg_search_module(Zycore QUIET IMPORTED_TARGET zycore)
find_package_handle_standard_args(Zycore
REQUIRED_VARS zycore_LINK_LIBRARIES
VERSION_VAR zycore_VERSION)
if (TARGET PkgConfig::zycore)
add_library(Zycore::Zycore ALIAS PkgConfig::zycore)
endif()
endif()
-23
View File
@@ -1,23 +0,0 @@
# SPDX-License-Identifier: MIT
if (CMAKE_CROSSCOMPILING)
return()
endif()
include(FindPackageHandleStandardArgs)
find_package(Zydis QUIET CONFIG)
if (Zydis_CONSIDERED_CONFIGS)
find_package_handle_standard_args(Zydis CONFIG_MODE)
else()
find_package(PkgConfig QUIET)
pkg_search_module(Zydis QUIET IMPORTED_TARGET zydis)
find_package_handle_standard_args(Zydis
REQUIRED_VARS zydis_LINK_LIBRARIES
VERSION_VAR zydis_VERSION)
if (TARGET PkgConfig::zydis)
add_library(Zydis::Zydis ALIAS PkgConfig::zydis)
endif()
endif()
-18
View File
@@ -1,18 +0,0 @@
# SPDX-License-Identifier: MIT
include(FindPackageHandleStandardArgs)
find_package(PkgConfig QUIET)
pkg_search_module(xxhash QUIET IMPORTED_TARGET xxhash libxxhash)
find_package_handle_standard_args(xxhash
REQUIRED_VARS xxhash_LINK_LIBRARIES
VERSION_VAR xxhash_VERSION
)
if (xxhash_FOUND AND NOT TARGET xxHash::xxhash)
if (TARGET PkgConfig::xxhash)
add_library(xxHash::xxhash ALIAS PkgConfig::xxhash)
else()
add_library(xxHash::xxhash ALIAS xxhash)
endif()
endif()
-15
View File
@@ -1,15 +0,0 @@
# SPDX-License-Identifier: MIT
# This applies some common linker options that reduce code size and linking time in Release mode. Namely:
# --gc-sections: Linktime garbage collection, discards unused sections from the final output
# --strip-all : Similar to running `strip`, discards the symbol table from the final output
# --as-needed : Only includes libraries that are actually needed in the final output.
macro(LinkerGC target)
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
target_link_options(${target} PRIVATE
"LINKER:--gc-sections"
"LINKER:--strip-all"
"LINKER:--as-needed")
endif()
endmacro()
-1
View File
@@ -4,7 +4,6 @@ set(CMAKE_RC_COMPILER ${MINGW_TRIPLE}-windres)
set(CMAKE_C_COMPILER ${MINGW_TRIPLE}-clang)
set(CMAKE_CXX_COMPILER ${MINGW_TRIPLE}-clang++)
set(CMAKE_DLLTOOL ${MINGW_TRIPLE}-dlltool)
set(CMAKE_AR ${MINGW_TRIPLE}-ar)
# Compile everything as static to avoid requiring the MinGW runtime libraries, force page aligned sections so that
# debug symbols work correctly, and disable loop alignment to workaround an LLVM bug
+1 -1
View File
@@ -14,7 +14,7 @@ RUN mkdir build
ARG CC=clang-13
ARG CXX=clang++-13
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTING=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTS=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN ninja
WORKDIR /FEX/build
+9 -7
View File
@@ -3,21 +3,23 @@ function(GenBinFmt Name)
get_filename_component(FMT_NAME ${Name} NAME_WE)
# Configure it
configure_file(${Name} ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
configure_file(
${Name}
${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
# Then install the configured binfmt
install(FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/
COMPONENT Runtime)
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
endfunction()
if (NOT USE_LEGACY_BINFMTMISC)
configure_file(FEX-x86.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf)
configure_file(FEX-x86_64.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf)
install(FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/
COMPONENT Runtime)
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
else()
GenBinFmt(FEX-x86.in)
GenBinFmt(FEX-x86_64.in)
+1 -1
View File
@@ -1 +1 @@
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+1 -1
View File
@@ -1 +1 @@
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+3 -3
View File
@@ -2,8 +2,8 @@
let
toolchain = pkgs.fetchzip {
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250920/llvm-mingw-20250920-ucrt-ubuntu-22.04-aarch64.tar.xz";
sha256 = "sha256-LaojKjC8KzY+soW5u6eoDoXE3qtYk9Ejr7M3enTqRAE=";
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250305/llvm-mingw-20250305-ucrt-ubuntu-20.04-aarch64.tar.xz";
sha256 = "sha256-cA03/ab9O61eO9+S2JzIXD4V0HzTXK5/AYyxW2d73Po=";
};
cmakeToolchainFile = pkgs.substitute {
@@ -45,7 +45,7 @@ pkgs.mkShell {
fi
'';
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False
FEX_CMAKE_TOOLCHAIN_ARM64EC = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_CMAKE_TOOLCHAIN_WOW64 = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_MESON_CROSSFILE = "--cross-file ${mesonCrossFile}";
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -14,4 +14,4 @@ fi
rm -rf unittests/FEXLinuxTests
set -o xtrace
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTING=ON -DBUILD_FEX_LINUX_TESTS=ON
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTS=ON -DBUILD_FEX_LINUX_TESTS=ON
-1
View File
@@ -1 +0,0 @@
DisableFormat: true
+1 -1
+3 -2
View File
@@ -1,5 +1,5 @@
add_library(softfloat_3e STATIC
set (SRCS
# F80 support
src/extF80_add.c
src/extF80_div.c
@@ -84,7 +84,7 @@ add_library(softfloat_3e STATIC
src/s_normSubnormalF32Sig.c
src/s_f32UIToCommonNaN.c)
if (ARCHITECTURE_arm64 AND HAS_CLANG_PRESERVE_ALL)
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=__attribute__((preserve_all));-DFEXCORE_HAS_PRESERVE_ALL_ATTR=1")
else()
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=;-DFEXCORE_HAS_PRESERVE_ALL_ATTR=0")
@@ -92,6 +92,7 @@ endif()
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ=1;-DINLINE=static inline;-DINLINE_LEVEL=4;-DSOFTFLOAT_FAST_INT64=1;-DSOFTFLOAT_FAST_DIV32TO16=1;-DSOFTFLOAT_FAST_DIV64TO32=1")
add_library(softfloat_3e STATIC ${SRCS})
target_include_directories(softfloat_3e PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include/)
target_include_directories(softfloat_3e PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include/SoftFloat-3e/)
target_compile_definitions(softfloat_3e PUBLIC ${DEFINES})
+2 -1
View File
@@ -1,4 +1,4 @@
add_library(cephes_128bit STATIC
set(SRCS_128BIT
src/128bit/Impl.cpp
src/128bit/atanll.c
src/128bit/constll.c
@@ -11,6 +11,7 @@ add_library(cephes_128bit STATIC
src/128bit/tanll.c)
# 128-bit library
add_library(cephes_128bit STATIC ${SRCS_128BIT})
target_link_libraries(cephes_128bit softfloat_3e)
target_include_directories(cephes_128bit PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include/)
target_compile_options(cephes_128bit PRIVATE -fno-builtin)
+8 -3
View File
@@ -169,9 +169,14 @@ View the diff from {self.name} here.
class ClangFormatHelper(FormatHelper):
name = "git-clang-format"
name = "clang-format"
friendly_name = "C/C++ code formatter"
@property
def cformat_wrapper_path(self) -> str:
relpath = "../../Scripts/clang-format.py"
curpath = os.path.dirname(os.path.abspath(__file__))
return os.path.abspath(os.path.normpath(os.path.join(curpath, relpath)))
@property
def instructions(self) -> str:
@@ -194,7 +199,7 @@ class ClangFormatHelper(FormatHelper):
def clang_fmt_path(self) -> str:
if "CLANG_FORMAT_PATH" in os.environ:
return os.environ["CLANG_FORMAT_PATH"]
return "git-clang-format-19"
return "git-clang-format"
def has_tool(self) -> bool:
cmd = [self.clang_fmt_path, "-h"]
@@ -212,7 +217,7 @@ class ClangFormatHelper(FormatHelper):
cf_cmd = [
self.clang_fmt_path,
"--binary=clang-format-19",
f"--binary={self.cformat_wrapper_path}",
"--diff",
]
+32 -421
View File
@@ -1,441 +1,52 @@
#
# This file is autogenerated by pip-compile with Python 3.13
# This file is autogenerated by pip-compile with Python 3.11
# by the following command:
#
# pip-compile --generate-hashes --output-file=requirements_formatting.txt --strip-extras requirements_formatting.txt.in
# pip-compile --output-file=llvm/utils/git/requirements_formatting.txt llvm/utils/git/requirements_formatting.txt.in
#
black==25.1.0 \
--hash=sha256:030b9759066a4ee5e5aca28c3c77f9c64789cdd4de8ac1df642c40b708be6171 \
--hash=sha256:055e59b198df7ac0b7efca5ad7ff2516bca343276c466be72eb04a3bcc1f82d7 \
--hash=sha256:0e519ecf93120f34243e6b0054db49c00a35f84f195d5bce7e9f5cfc578fc2da \
--hash=sha256:172b1dbff09f86ce6f4eb8edf9dede08b1fce58ba194c87d7a4f1a5aa2f5b3c2 \
--hash=sha256:1e2978f6df243b155ef5fa7e558a43037c3079093ed5d10fd84c43900f2d8ecc \
--hash=sha256:33496d5cd1222ad73391352b4ae8da15253c5de89b93a80b3e2c8d9a19ec2666 \
--hash=sha256:3b48735872ec535027d979e8dcb20bf4f70b5ac75a8ea99f127c106a7d7aba9f \
--hash=sha256:4b60580e829091e6f9238c848ea6750efed72140b91b048770b64e74fe04908b \
--hash=sha256:759e7ec1e050a15f89b770cefbf91ebee8917aac5c20483bc2d80a6c3a04df32 \
--hash=sha256:8f0b18a02996a836cc9c9c78e5babec10930862827b1b724ddfe98ccf2f2fe4f \
--hash=sha256:95e8176dae143ba9097f351d174fdaf0ccd29efb414b362ae3fd72bf0f710717 \
--hash=sha256:96c1c7cd856bba8e20094e36e0f948718dc688dba4a9d78c3adde52b9e6c2299 \
--hash=sha256:a1ee0a0c330f7b5130ce0caed9936a904793576ef4d2b98c40835d6a65afa6a0 \
--hash=sha256:a22f402b410566e2d1c950708c77ebf5ebd5d0d88a6a2e87c86d9fb48afa0d18 \
--hash=sha256:a39337598244de4bae26475f77dda852ea00a93bd4c728e09eacd827ec929df0 \
--hash=sha256:afebb7098bfbc70037a053b91ae8437c3857482d3a690fefc03e9ff7aa9a5fd3 \
--hash=sha256:bacabb307dca5ebaf9c118d2d2f6903da0d62c9faa82bd21a33eecc319559355 \
--hash=sha256:bce2e264d59c91e52d8000d507eb20a9aca4a778731a08cfff7e5ac4a4bb7096 \
--hash=sha256:d9e6827d563a2c820772b32ce8a42828dc6790f095f441beef18f96aa6f8294e \
--hash=sha256:db8ea9917d6f8fc62abd90d944920d95e73c83a5ee3383493e35d271aca872e9 \
--hash=sha256:ea0213189960bda9cf99be5b8c8ce66bb054af5e9e861249cd23471bd7b0b3ba \
--hash=sha256:f3df5f1bf91d36002b0a75389ca8663510cf0531cca8aa5c1ef695b46d98655f
black==23.9.1
# via
# -r requirements_formatting.txt.in
# -r llvm/utils/git/requirements_formatting.txt.in
# darker
certifi==2025.7.14 \
--hash=sha256:6b31f564a415d79ee77df69d757bb49a5bb53bd9f756cbbe24394ffd6fc1f4b2 \
--hash=sha256:8ea99dbdfaaf2ba2f9bac77b9249ef62ec5218e7c2b2e903378ed5fccf765995
# via
# -r requirements_formatting.txt.in
# requests
cffi==2.0.0 \
--hash=sha256:00bdf7acc5f795150faa6957054fbbca2439db2f775ce831222b66f192f03beb \
--hash=sha256:07b271772c100085dd28b74fa0cd81c8fb1a3ba18b21e03d7c27f3436a10606b \
--hash=sha256:087067fa8953339c723661eda6b54bc98c5625757ea62e95eb4898ad5e776e9f \
--hash=sha256:0a1527a803f0a659de1af2e1fd700213caba79377e27e4693648c2923da066f9 \
--hash=sha256:0cf2d91ecc3fcc0625c2c530fe004f82c110405f101548512cce44322fa8ac44 \
--hash=sha256:0f6084a0ea23d05d20c3edcda20c3d006f9b6f3fefeac38f59262e10cef47ee2 \
--hash=sha256:12873ca6cb9b0f0d3a0da705d6086fe911591737a59f28b7936bdfed27c0d47c \
--hash=sha256:19f705ada2530c1167abacb171925dd886168931e0a7b78f5bffcae5c6b5be75 \
--hash=sha256:1cd13c99ce269b3ed80b417dcd591415d3372bcac067009b6e0f59c7d4015e65 \
--hash=sha256:1e3a615586f05fc4065a8b22b8152f0c1b00cdbc60596d187c2a74f9e3036e4e \
--hash=sha256:1f72fb8906754ac8a2cc3f9f5aaa298070652a0ffae577e0ea9bd480dc3c931a \
--hash=sha256:1fc9ea04857caf665289b7a75923f2c6ed559b8298a1b8c49e59f7dd95c8481e \
--hash=sha256:203a48d1fb583fc7d78a4c6655692963b860a417c0528492a6bc21f1aaefab25 \
--hash=sha256:2081580ebb843f759b9f617314a24ed5738c51d2aee65d31e02f6f7a2b97707a \
--hash=sha256:21d1152871b019407d8ac3985f6775c079416c282e431a4da6afe7aefd2bccbe \
--hash=sha256:24b6f81f1983e6df8db3adc38562c83f7d4a0c36162885ec7f7b77c7dcbec97b \
--hash=sha256:256f80b80ca3853f90c21b23ee78cd008713787b1b1e93eae9f3d6a7134abd91 \
--hash=sha256:28a3a209b96630bca57cce802da70c266eb08c6e97e5afd61a75611ee6c64592 \
--hash=sha256:2c8f814d84194c9ea681642fd164267891702542f028a15fc97d4674b6206187 \
--hash=sha256:2de9a304e27f7596cd03d16f1b7c72219bd944e99cc52b84d0145aefb07cbd3c \
--hash=sha256:38100abb9d1b1435bc4cc340bb4489635dc2f0da7456590877030c9b3d40b0c1 \
--hash=sha256:3925dd22fa2b7699ed2617149842d2e6adde22b262fcbfada50e3d195e4b3a94 \
--hash=sha256:3e17ed538242334bf70832644a32a7aae3d83b57567f9fd60a26257e992b79ba \
--hash=sha256:3e837e369566884707ddaf85fc1744b47575005c0a229de3327f8f9a20f4efeb \
--hash=sha256:3f4d46d8b35698056ec29bca21546e1551a205058ae1a181d871e278b0b28165 \
--hash=sha256:44d1b5909021139fe36001ae048dbdde8214afa20200eda0f64c068cac5d5529 \
--hash=sha256:45d5e886156860dc35862657e1494b9bae8dfa63bf56796f2fb56e1679fc0bca \
--hash=sha256:4647afc2f90d1ddd33441e5b0e85b16b12ddec4fca55f0d9671fef036ecca27c \
--hash=sha256:4671d9dd5ec934cb9a73e7ee9676f9362aba54f7f34910956b84d727b0d73fb6 \
--hash=sha256:53f77cbe57044e88bbd5ed26ac1d0514d2acf0591dd6bb02a3ae37f76811b80c \
--hash=sha256:5eda85d6d1879e692d546a078b44251cdd08dd1cfb98dfb77b670c97cee49ea0 \
--hash=sha256:5fed36fccc0612a53f1d4d9a816b50a36702c28a2aa880cb8a122b3466638743 \
--hash=sha256:61d028e90346df14fedc3d1e5441df818d095f3b87d286825dfcbd6459b7ef63 \
--hash=sha256:66f011380d0e49ed280c789fbd08ff0d40968ee7b665575489afa95c98196ab5 \
--hash=sha256:6824f87845e3396029f3820c206e459ccc91760e8fa24422f8b0c3d1731cbec5 \
--hash=sha256:6c6c373cfc5c83a975506110d17457138c8c63016b563cc9ed6e056a82f13ce4 \
--hash=sha256:6d02d6655b0e54f54c4ef0b94eb6be0607b70853c45ce98bd278dc7de718be5d \
--hash=sha256:6d50360be4546678fc1b79ffe7a66265e28667840010348dd69a314145807a1b \
--hash=sha256:730cacb21e1bdff3ce90babf007d0a0917cc3e6492f336c2f0134101e0944f93 \
--hash=sha256:737fe7d37e1a1bffe70bd5754ea763a62a066dc5913ca57e957824b72a85e205 \
--hash=sha256:74a03b9698e198d47562765773b4a8309919089150a0bb17d829ad7b44b60d27 \
--hash=sha256:7553fb2090d71822f02c629afe6042c299edf91ba1bf94951165613553984512 \
--hash=sha256:7a66c7204d8869299919db4d5069a82f1561581af12b11b3c9f48c584eb8743d \
--hash=sha256:7cc09976e8b56f8cebd752f7113ad07752461f48a58cbba644139015ac24954c \
--hash=sha256:81afed14892743bbe14dacb9e36d9e0e504cd204e0b165062c488942b9718037 \
--hash=sha256:8941aaadaf67246224cee8c3803777eed332a19d909b47e29c9842ef1e79ac26 \
--hash=sha256:89472c9762729b5ae1ad974b777416bfda4ac5642423fa93bd57a09204712322 \
--hash=sha256:8ea985900c5c95ce9db1745f7933eeef5d314f0565b27625d9a10ec9881e1bfb \
--hash=sha256:8eca2a813c1cb7ad4fb74d368c2ffbbb4789d377ee5bb8df98373c2cc0dee76c \
--hash=sha256:92b68146a71df78564e4ef48af17551a5ddd142e5190cdf2c5624d0c3ff5b2e8 \
--hash=sha256:9332088d75dc3241c702d852d4671613136d90fa6881da7d770a483fd05248b4 \
--hash=sha256:94698a9c5f91f9d138526b48fe26a199609544591f859c870d477351dc7b2414 \
--hash=sha256:9a67fc9e8eb39039280526379fb3a70023d77caec1852002b4da7e8b270c4dd9 \
--hash=sha256:9de40a7b0323d889cf8d23d1ef214f565ab154443c42737dfe52ff82cf857664 \
--hash=sha256:a05d0c237b3349096d3981b727493e22147f934b20f6f125a3eba8f994bec4a9 \
--hash=sha256:afb8db5439b81cf9c9d0c80404b60c3cc9c3add93e114dcae767f1477cb53775 \
--hash=sha256:b18a3ed7d5b3bd8d9ef7a8cb226502c6bf8308df1525e1cc676c3680e7176739 \
--hash=sha256:b1e74d11748e7e98e2f426ab176d4ed720a64412b6a15054378afdb71e0f37dc \
--hash=sha256:b21e08af67b8a103c71a250401c78d5e0893beff75e28c53c98f4de42f774062 \
--hash=sha256:b4c854ef3adc177950a8dfc81a86f5115d2abd545751a304c5bcf2c2c7283cfe \
--hash=sha256:b882b3df248017dba09d6b16defe9b5c407fe32fc7c65a9c69798e6175601be9 \
--hash=sha256:baf5215e0ab74c16e2dd324e8ec067ef59e41125d3eade2b863d294fd5035c92 \
--hash=sha256:c649e3a33450ec82378822b3dad03cc228b8f5963c0c12fc3b1e0ab940f768a5 \
--hash=sha256:c654de545946e0db659b3400168c9ad31b5d29593291482c43e3564effbcee13 \
--hash=sha256:c6638687455baf640e37344fe26d37c404db8b80d037c3d29f58fe8d1c3b194d \
--hash=sha256:c8d3b5532fc71b7a77c09192b4a5a200ea992702734a2e9279a37f2478236f26 \
--hash=sha256:cb527a79772e5ef98fb1d700678fe031e353e765d1ca2d409c92263c6d43e09f \
--hash=sha256:cf364028c016c03078a23b503f02058f1814320a56ad535686f90565636a9495 \
--hash=sha256:d48a880098c96020b02d5a1f7d9251308510ce8858940e6fa99ece33f610838b \
--hash=sha256:d68b6cef7827e8641e8ef16f4494edda8b36104d79773a334beaa1e3521430f6 \
--hash=sha256:d9b29c1f0ae438d5ee9acb31cadee00a58c46cc9c0b2f9038c6b0b3470877a8c \
--hash=sha256:d9b97165e8aed9272a6bb17c01e3cc5871a594a446ebedc996e2397a1c1ea8ef \
--hash=sha256:da68248800ad6320861f129cd9c1bf96ca849a2771a59e0344e88681905916f5 \
--hash=sha256:da902562c3e9c550df360bfa53c035b2f241fed6d9aef119048073680ace4a18 \
--hash=sha256:dbd5c7a25a7cb98f5ca55d258b103a2054f859a46ae11aaf23134f9cc0d356ad \
--hash=sha256:dd4f05f54a52fb558f1ba9f528228066954fee3ebe629fc1660d874d040ae5a3 \
--hash=sha256:de8dad4425a6ca6e4e5e297b27b5c824ecc7581910bf9aee86cb6835e6812aa7 \
--hash=sha256:e11e82b744887154b182fd3e7e8512418446501191994dbf9c9fc1f32cc8efd5 \
--hash=sha256:e6e73b9e02893c764e7e8d5bb5ce277f1a009cd5243f8228f75f842bf937c534 \
--hash=sha256:f73b96c41e3b2adedc34a7356e64c8eb96e03a3782b535e043a986276ce12a49 \
--hash=sha256:f93fd8e5c8c0a4aa1f424d6173f14a892044054871c771f8566e4008eaa359d2 \
--hash=sha256:fc33c5141b55ed366cfaad382df24fe7dcbc686de5be719b207bb248e3053dc5 \
--hash=sha256:fc7de24befaeae77ba923797c7c87834c73648a05a4bde34b3b7e5588973a453 \
--hash=sha256:fe562eb1a64e67dd297ccc4f5addea2501664954f2692b69a76449ec7913ecbf
certifi==2023.7.22
# via requests
cffi==1.15.1
# via
# cryptography
# pynacl
charset-normalizer==3.2.0 \
--hash=sha256:04e57ab9fbf9607b77f7d057974694b4f6b142da9ed4a199859d9d4d5c63fe96 \
--hash=sha256:09393e1b2a9461950b1c9a45d5fd251dc7c6f228acab64da1c9c0165d9c7765c \
--hash=sha256:0b87549028f680ca955556e3bd57013ab47474c3124dc069faa0b6545b6c9710 \
--hash=sha256:1000fba1057b92a65daec275aec30586c3de2401ccdcd41f8a5c1e2c87078706 \
--hash=sha256:1249cbbf3d3b04902ff081ffbb33ce3377fa6e4c7356f759f3cd076cc138d020 \
--hash=sha256:1920d4ff15ce893210c1f0c0e9d19bfbecb7983c76b33f046c13a8ffbd570252 \
--hash=sha256:193cbc708ea3aca45e7221ae58f0fd63f933753a9bfb498a3b474878f12caaad \
--hash=sha256:1a100c6d595a7f316f1b6f01d20815d916e75ff98c27a01ae817439ea7726329 \
--hash=sha256:1f30b48dd7fa1474554b0b0f3fdfdd4c13b5c737a3c6284d3cdc424ec0ffff3a \
--hash=sha256:203f0c8871d5a7987be20c72442488a0b8cfd0f43b7973771640fc593f56321f \
--hash=sha256:246de67b99b6851627d945db38147d1b209a899311b1305dd84916f2b88526c6 \
--hash=sha256:2dee8e57f052ef5353cf608e0b4c871aee320dd1b87d351c28764fc0ca55f9f4 \
--hash=sha256:2efb1bd13885392adfda4614c33d3b68dee4921fd0ac1d3988f8cbb7d589e72a \
--hash=sha256:2f4ac36d8e2b4cc1aa71df3dd84ff8efbe3bfb97ac41242fbcfc053c67434f46 \
--hash=sha256:3170c9399da12c9dc66366e9d14da8bf7147e1e9d9ea566067bbce7bb74bd9c2 \
--hash=sha256:3b1613dd5aee995ec6d4c69f00378bbd07614702a315a2cf6c1d21461fe17c23 \
--hash=sha256:3bb3d25a8e6c0aedd251753a79ae98a093c7e7b471faa3aa9a93a81431987ace \
--hash=sha256:3bb7fda7260735efe66d5107fb7e6af6a7c04c7fce9b2514e04b7a74b06bf5dd \
--hash=sha256:41b25eaa7d15909cf3ac4c96088c1f266a9a93ec44f87f1d13d4a0e86c81b982 \
--hash=sha256:45de3f87179c1823e6d9e32156fb14c1927fcc9aba21433f088fdfb555b77c10 \
--hash=sha256:46fb8c61d794b78ec7134a715a3e564aafc8f6b5e338417cb19fe9f57a5a9bf2 \
--hash=sha256:48021783bdf96e3d6de03a6e39a1171ed5bd7e8bb93fc84cc649d11490f87cea \
--hash=sha256:4957669ef390f0e6719db3613ab3a7631e68424604a7b448f079bee145da6e09 \
--hash=sha256:5e86d77b090dbddbe78867a0275cb4df08ea195e660f1f7f13435a4649e954e5 \
--hash=sha256:6339d047dab2780cc6220f46306628e04d9750f02f983ddb37439ca47ced7149 \
--hash=sha256:681eb3d7e02e3c3655d1b16059fbfb605ac464c834a0c629048a30fad2b27489 \
--hash=sha256:6c409c0deba34f147f77efaa67b8e4bb83d2f11c8806405f76397ae5b8c0d1c9 \
--hash=sha256:7095f6fbfaa55defb6b733cfeb14efaae7a29f0b59d8cf213be4e7ca0b857b80 \
--hash=sha256:70c610f6cbe4b9fce272c407dd9d07e33e6bf7b4aa1b7ffb6f6ded8e634e3592 \
--hash=sha256:72814c01533f51d68702802d74f77ea026b5ec52793c791e2da806a3844a46c3 \
--hash=sha256:7a4826ad2bd6b07ca615c74ab91f32f6c96d08f6fcc3902ceeedaec8cdc3bcd6 \
--hash=sha256:7c70087bfee18a42b4040bb9ec1ca15a08242cf5867c58726530bdf3945672ed \
--hash=sha256:855eafa5d5a2034b4621c74925d89c5efef61418570e5ef9b37717d9c796419c \
--hash=sha256:8700f06d0ce6f128de3ccdbc1acaea1ee264d2caa9ca05daaf492fde7c2a7200 \
--hash=sha256:89f1b185a01fe560bc8ae5f619e924407efca2191b56ce749ec84982fc59a32a \
--hash=sha256:8b2c760cfc7042b27ebdb4a43a4453bd829a5742503599144d54a032c5dc7e9e \
--hash=sha256:8c2f5e83493748286002f9369f3e6607c565a6a90425a3a1fef5ae32a36d749d \
--hash=sha256:8e098148dd37b4ce3baca71fb394c81dc5d9c7728c95df695d2dca218edf40e6 \
--hash=sha256:94aea8eff76ee6d1cdacb07dd2123a68283cb5569e0250feab1240058f53b623 \
--hash=sha256:95eb302ff792e12aba9a8b8f8474ab229a83c103d74a750ec0bd1c1eea32e669 \
--hash=sha256:9bd9b3b31adcb054116447ea22caa61a285d92e94d710aa5ec97992ff5eb7cf3 \
--hash=sha256:9e608aafdb55eb9f255034709e20d5a83b6d60c054df0802fa9c9883d0a937aa \
--hash=sha256:a103b3a7069b62f5d4890ae1b8f0597618f628b286b03d4bc9195230b154bfa9 \
--hash=sha256:a386ebe437176aab38c041de1260cd3ea459c6ce5263594399880bbc398225b2 \
--hash=sha256:a38856a971c602f98472050165cea2cdc97709240373041b69030be15047691f \
--hash=sha256:a401b4598e5d3f4a9a811f3daf42ee2291790c7f9d74b18d75d6e21dda98a1a1 \
--hash=sha256:a7647ebdfb9682b7bb97e2a5e7cb6ae735b1c25008a70b906aecca294ee96cf4 \
--hash=sha256:aaf63899c94de41fe3cf934601b0f7ccb6b428c6e4eeb80da72c58eab077b19a \
--hash=sha256:b0dac0ff919ba34d4df1b6131f59ce95b08b9065233446be7e459f95554c0dc8 \
--hash=sha256:baacc6aee0b2ef6f3d308e197b5d7a81c0e70b06beae1f1fcacffdbd124fe0e3 \
--hash=sha256:bf420121d4c8dce6b889f0e8e4ec0ca34b7f40186203f06a946fa0276ba54029 \
--hash=sha256:c04a46716adde8d927adb9457bbe39cf473e1e2c2f5d0a16ceb837e5d841ad4f \
--hash=sha256:c0b21078a4b56965e2b12f247467b234734491897e99c1d51cee628da9786959 \
--hash=sha256:c1c76a1743432b4b60ab3358c937a3fe1341c828ae6194108a94c69028247f22 \
--hash=sha256:c4983bf937209c57240cff65906b18bb35e64ae872da6a0db937d7b4af845dd7 \
--hash=sha256:c4fb39a81950ec280984b3a44f5bd12819953dc5fa3a7e6fa7a80db5ee853952 \
--hash=sha256:c57921cda3a80d0f2b8aec7e25c8aa14479ea92b5b51b6876d975d925a2ea346 \
--hash=sha256:c8063cf17b19661471ecbdb3df1c84f24ad2e389e326ccaf89e3fb2484d8dd7e \
--hash=sha256:ccd16eb18a849fd8dcb23e23380e2f0a354e8daa0c984b8a732d9cfaba3a776d \
--hash=sha256:cd6dbe0238f7743d0efe563ab46294f54f9bc8f4b9bcf57c3c666cc5bc9d1299 \
--hash=sha256:d62e51710986674142526ab9f78663ca2b0726066ae26b78b22e0f5e571238dd \
--hash=sha256:db901e2ac34c931d73054d9797383d0f8009991e723dab15109740a63e7f902a \
--hash=sha256:e03b8895a6990c9ab2cdcd0f2fe44088ca1c65ae592b8f795c3294af00a461c3 \
--hash=sha256:e1c8a2f4c69e08e89632defbfabec2feb8a8d99edc9f89ce33c4b9e36ab63037 \
--hash=sha256:e4b749b9cc6ee664a3300bb3a273c1ca8068c46be705b6c31cf5d276f8628a94 \
--hash=sha256:e6a5bf2cba5ae1bb80b154ed68a3cfa2fa00fde979a7f50d6598d3e17d9ac20c \
--hash=sha256:e857a2232ba53ae940d3456f7533ce6ca98b81917d47adc3c7fd55dad8fab858 \
--hash=sha256:ee4006268ed33370957f55bf2e6f4d263eaf4dc3cfc473d1d90baff6ed36ce4a \
--hash=sha256:eef9df1eefada2c09a5e7a40991b9fc6ac6ef20b1372abd48d2794a316dc0449 \
--hash=sha256:f058f6963fd82eb143c692cecdc89e075fa0828db2e5b291070485390b2f1c9c \
--hash=sha256:f25c229a6ba38a35ae6e25ca1264621cc25d4d38dca2942a7fce0b67a4efe918 \
--hash=sha256:f2a1d0fd4242bd8643ce6f98927cf9c04540af6efa92323e9d3124f57727bfc1 \
--hash=sha256:f7560358a6811e52e9c4d142d497f1a6e10103d3a6881f18d04dbce3729c0e2c \
--hash=sha256:f779d3ad205f108d14e99bb3859aa7dd8e9c68874617c72354d7ecaec2a054ac \
--hash=sha256:f87f746ee241d30d6ed93969de31e5ffd09a2961a051e60ae6bddde9ec3583aa
charset-normalizer==3.2.0
# via requests
click==8.1.7 \
--hash=sha256:ae74fb96c20a0277a1d615f1e4d73c8414f5a98db8b799a7931d1582f3390c28 \
--hash=sha256:ca9853ad459e787e2192211578cc907e7594e294c7ccc834310722b41b9ca6de
click==8.1.7
# via black
cryptography==46.0.5 \
--hash=sha256:02f547fce831f5096c9a567fd41bc12ca8f11df260959ecc7c3202555cc47a72 \
--hash=sha256:039917b0dc418bb9f6edce8a906572d69e74bd330b0b3fea4f79dab7f8ddd235 \
--hash=sha256:1abfdb89b41c3be0365328a410baa9df3ff8a9110fb75e7b52e66803ddabc9a9 \
--hash=sha256:2ae6971afd6246710480e3f15824ed3029a60fc16991db250034efd0b9fb4356 \
--hash=sha256:2b7a67c9cd56372f3249b39699f2ad479f6991e62ea15800973b956f4b73e257 \
--hash=sha256:351695ada9ea9618b3500b490ad54c739860883df6c1f555e088eaf25b1bbaad \
--hash=sha256:38946c54b16c885c72c4f59846be9743d699eee2b69b6988e0a00a01f46a61a4 \
--hash=sha256:3b4995dc971c9fb83c25aa44cf45f02ba86f71ee600d81091c2f0cbae116b06c \
--hash=sha256:3ce58ba46e1bc2aac4f7d9290223cead56743fa6ab94a5d53292ffaac6a91614 \
--hash=sha256:3ee190460e2fbe447175cda91b88b84ae8322a104fc27766ad09428754a618ed \
--hash=sha256:4108d4c09fbbf2789d0c926eb4152ae1760d5a2d97612b92d508d96c861e4d31 \
--hash=sha256:420d0e909050490d04359e7fdb5ed7e667ca5c3c402b809ae2563d7e66a92229 \
--hash=sha256:47fb8a66058b80e509c47118ef8a75d14c455e81ac369050f20ba0d23e77fee0 \
--hash=sha256:4c3341037c136030cb46e4b1e17b7418ea4cbd9dd207e4a6f3b2b24e0d4ac731 \
--hash=sha256:4d7e3d356b8cd4ea5aff04f129d5f66ebdc7b6f8eae802b93739ed520c47c79b \
--hash=sha256:4d8ae8659ab18c65ced284993c2265910f6c9e650189d4e3f68445ef82a810e4 \
--hash=sha256:4e817a8920bfbcff8940ecfd60f23d01836408242b30f1a708d93198393a80b4 \
--hash=sha256:50bfb6925eff619c9c023b967d5b77a54e04256c4281b0e21336a130cd7fc263 \
--hash=sha256:556e106ee01aa13484ce9b0239bca667be5004efb0aabbed28d353df86445595 \
--hash=sha256:582f5fcd2afa31622f317f80426a027f30dc792e9c80ffee87b993200ea115f1 \
--hash=sha256:5be7bf2fb40769e05739dd0046e7b26f9d4670badc7b032d6ce4db64dddc0678 \
--hash=sha256:60ee7e19e95104d4c03871d7d7dfb3d22ef8a9b9c6778c94e1c8fcc8365afd48 \
--hash=sha256:61aa400dce22cb001a98014f647dc21cda08f7915ceb95df0c9eaf84b4b6af76 \
--hash=sha256:68f68d13f2e1cb95163fa3b4db4bf9a159a418f5f6e7242564fc75fcae667fd0 \
--hash=sha256:7d1f30a86d2757199cb2d56e48cce14deddf1f9c95f1ef1b64ee91ea43fe2e18 \
--hash=sha256:7d731d4b107030987fd61a7f8ab512b25b53cef8f233a97379ede116f30eb67d \
--hash=sha256:803812e111e75d1aa73690d2facc295eaefd4439be1023fefc4995eaea2af90d \
--hash=sha256:80a8d7bfdf38f87ca30a5391c0c9ce4ed2926918e017c29ddf643d0ed2778ea1 \
--hash=sha256:8293f3dea7fc929ef7240796ba231413afa7b68ce38fd21da2995549f5961981 \
--hash=sha256:8456928655f856c6e1533ff59d5be76578a7157224dbd9ce6872f25055ab9ab7 \
--hash=sha256:890bcb4abd5a2d3f852196437129eb3667d62630333aacc13dfd470fad3aaa82 \
--hash=sha256:94a76daa32eb78d61339aff7952ea819b1734b46f73646a07decb40e5b3448e2 \
--hash=sha256:9f16fbdf4da055efb21c22d81b89f155f02ba420558db21288b3d0035bafd5f4 \
--hash=sha256:a3d1fae9863299076f05cb8a778c467578262fae09f9dc0ee9b12eb4268ce663 \
--hash=sha256:a3d507bb6a513ca96ba84443226af944b0f7f47dcc9a399d110cd6146481d24c \
--hash=sha256:abace499247268e3757271b2f1e244b36b06f8515cf27c4d49468fc9eb16e93d \
--hash=sha256:ba2a27ff02f48193fc4daeadf8ad2590516fa3d0adeeb34336b96f7fa64c1e3a \
--hash=sha256:bc84e875994c3b445871ea7181d424588171efec3e185dced958dad9e001950a \
--hash=sha256:bfd56bb4b37ed4f330b82402f6f435845a5f5648edf1ad497da51a8452d5d62d \
--hash=sha256:c18ff11e86df2e28854939acde2d003f7984f721eba450b56a200ad90eeb0e6b \
--hash=sha256:c3bcce8521d785d510b2aad26ae2c966092b7daa8f45dd8f44734a104dc0bc1a \
--hash=sha256:c4143987a42a2397f2fc3b4d7e3a7d313fbe684f67ff443999e803dd75a76826 \
--hash=sha256:c69fd885df7d089548a42d5ec05be26050ebcd2283d89b3d30676eb32ff87dee \
--hash=sha256:ced80795227d70549a411a4ab66e8ce307899fad2220ce5ab2f296e687eacde9 \
--hash=sha256:d66e421495fdb797610a08f43b05269e0a5ea7f5e652a89bfd5a7d3c1dee3648 \
--hash=sha256:d861ee9e76ace6cf36a6a89b959ec08e7bc2493ee39d07ffe5acb23ef46d27da \
--hash=sha256:e9251e3be159d1020c4030bd2e5f84d6a43fe54b6c19c12f51cde9542a2817b2 \
--hash=sha256:f145bba11b878005c496e93e257c1e88f154d278d2638e6450d17e0f31e558d2 \
--hash=sha256:fe346b143ff9685e40192a4960938545c699054ba11d4f9029f94751e3f71d87
# via
# -r requirements_formatting.txt.in
# pyjwt
darker==2.1.1 \
--hash=sha256:a6e6a682c0604e76fe9aec7650e96a944f517563c69b28fcc076db9d957d98ea \
--hash=sha256:ead701414c45359fc0312bc285614d3285fc135476d43f3bc08d989ee19d9020
# via -r requirements_formatting.txt.in
darkgraylib==1.2.1 \
--hash=sha256:60c59de69842367ce0c78c32c451fa8e9d29500e681312d9864a7416bcdb7792 \
--hash=sha256:a5dd6a2015a470d9047278cdd01a91ccb1d746675f8fd4562b3b5f6b8cbda930
# via
# darker
# graylint
deprecated==1.2.14 \
--hash=sha256:6fac8b097794a90302bdbb17b9b815e732d3c4720583ff1b198499d78470466c \
--hash=sha256:e5323eb936458dccc2582dc6f9c322c852a775a27065ff2b0c4970b9d53d01b3
cryptography==41.0.3
# via pyjwt
darker==1.7.2
# via -r llvm/utils/git/requirements_formatting.txt.in
deprecated==1.2.14
# via pygithub
graylint==1.1.1 \
--hash=sha256:0fd8e02972ca03d0ef2bf0adea76b5343efcd492d7afb5f658f3e3a724f55a36 \
--hash=sha256:b7e0eab6c159684dbf5ef84e942c3340f6a6549b02a3d11b1a1763cc4f8f0593
# via darker
idna==3.10 \
--hash=sha256:12f65c9b470abda6dc35cf8e63cc574b1c52b11df2c86030af0ac09b01b13ea9 \
--hash=sha256:946d195a0d259cbba61165e88e65941f16e9b36ea6ddb97f00452bae8b1287d3
# via
# -r requirements_formatting.txt.in
# requests
mypy-extensions==1.0.0 \
--hash=sha256:4392f6c0eb8a5668a69e23d168ffa70f0be9ccfd32b5cc2d26a34ae5b844552d \
--hash=sha256:75dbf8955dc00442a438fc4d0666508a9a97b6bd41aa2f0ffe9d2f2725af0782
idna==3.4
# via requests
mypy-extensions==1.0.0
# via black
packaging==23.1 \
--hash=sha256:994793af429502c4ea2ebf6bf664629d07c1a9fe974af92966e4b8d2df7edc61 \
--hash=sha256:a392980d2b6cffa644431898be54b0045151319d1e7ec34f0cfed48767dd334f
packaging==23.1
# via black
pathspec==0.11.2 \
--hash=sha256:1d6ed233af05e679efb96b1851550ea95bbb64b7c490b0f5aa52996c11e92a20 \
--hash=sha256:e0d8d0ac2f12da61956eb2306b69f9469b42f4deb0f3cb6ed47b9cce9996ced3
pathspec==0.11.2
# via black
platformdirs==3.10.0 \
--hash=sha256:b45696dab2d7cc691a3226759c0d3b00c47c8b6e293d96f6436f733303f77f6d \
--hash=sha256:d7c24979f292f916dc9cbf8648319032f551ea8c49a4c9bf2fb556a02070ec1d
platformdirs==3.10.0
# via black
pycparser==2.21 \
--hash=sha256:8ee45429555515e1f6b185e78100aea234072576aa43ab53aefcae078162fca9 \
--hash=sha256:e644fdec12f7872f86c58ff790da456218b10f863970249516d60a5eaca77206
pycparser==2.21
# via cffi
pygithub==2.6.1 \
--hash=sha256:6f2fa6d076ccae475f9fc392cc6cdbd54db985d4f69b8833a28397de75ed6ca3 \
--hash=sha256:b5c035392991cca63959e9453286b41b54d83bf2de2daa7d7ff7e4312cebf3bf
# via -r requirements_formatting.txt.in
pyjwt==2.8.0 \
--hash=sha256:57e28d156e3d5c10088e0c68abb90bfac3df82b40a71bd0daa20c65ccd5c23de \
--hash=sha256:59127c392cc44c2da5bb3192169a91f429924e17aff6534d70fdc02ab3e04320
pygithub==1.59.1
# via -r llvm/utils/git/requirements_formatting.txt.in
pyjwt[crypto]==2.8.0
# via pygithub
pynacl==1.6.2 \
--hash=sha256:018494d6d696ae03c7e656e5e74cdfd8ea1326962cc401bcf018f1ed8436811c \
--hash=sha256:04316d1fc625d860b6c162fff704eb8426b1a8bcd3abacea11142cbd99a6b574 \
--hash=sha256:22de65bb9010a725b0dac248f353bb072969c94fa8d6b1f34b87d7953cf7bbe4 \
--hash=sha256:26bfcd00dcf2cf160f122186af731ae30ab120c18e8375684ec2670dccd28130 \
--hash=sha256:2fef529ef3ee487ad8113d287a593fa26f48ee3620d92ecc6f1d09ea38e0709b \
--hash=sha256:320ef68a41c87547c91a8b58903c9caa641ab01e8512ce291085b5fe2fcb7590 \
--hash=sha256:3bffb6d0f6becacb6526f8f42adfb5efb26337056ee0831fb9a7044d1a964444 \
--hash=sha256:44081faff368d6c5553ccf55322ef2819abb40e25afaec7e740f159f74813634 \
--hash=sha256:46065496ab748469cdd999246d17e301b2c24ae2fdf739132e580a0e94c94a87 \
--hash=sha256:5811c72b473b2f38f7e2a3dc4f8642e3a3e9b5e7317266e4ced1fba85cae41aa \
--hash=sha256:622d7b07cc5c02c666795792931b50c91f3ce3c2649762efb1ef0d5684c81594 \
--hash=sha256:62985f233210dee6548c223301b6c25440852e13d59a8b81490203c3227c5ba0 \
--hash=sha256:68be3a09455743ff9505491220b64440ced8973fe930f270c8e07ccfa25b1f9e \
--hash=sha256:834a43af110f743a754448463e8fd61259cd4ab5bbedcf70f9dabad1d28a394c \
--hash=sha256:8845c0631c0be43abdd865511c41eab235e0be69c81dc66a50911594198679b0 \
--hash=sha256:8a66d6fb6ae7661c58995f9c6435bda2b1e68b54b598a6a10247bfcdadac996c \
--hash=sha256:8b097553b380236d51ed11356c953bf8ce36a29a3e596e934ecabe76c985a577 \
--hash=sha256:a84bf1c20339d06dc0c85d9aea9637a24f718f375d861b2668b2f9f96fa51145 \
--hash=sha256:a9f9932d8d2811ce1a8ffa79dcbdf3970e7355b5c8eb0c1a881a57e7f7d96e88 \
--hash=sha256:bc4a36b28dd72fb4845e5d8f9760610588a96d5a51f01d84d8c6ff9849968c14 \
--hash=sha256:c8a231e36ec2cab018c4ad4358c386e36eede0319a0c41fed24f840b1dac59f6 \
--hash=sha256:c949ea47e4206af7c8f604b8278093b674f7c79ed0d4719cc836902bf4517465 \
--hash=sha256:d071c6a9a4c94d79eb665db4ce5cedc537faf74f2355e4d502591d850d3913c0 \
--hash=sha256:d29bfe37e20e015a7d8b23cfc8bd6aa7909c92a1b8f41ee416bbb3e79ef182b2 \
--hash=sha256:fe9847ca47d287af41e82be1dd5e23023d3c31a951da134121ab02e42ac218c9
# via
# -r requirements_formatting.txt.in
# pygithub
requests==2.32.4 \
--hash=sha256:27babd3cda2a6d50b30443204ee89830707d396671944c998b5975b031ac2b2c \
--hash=sha256:27d0316682c8a29834d3264820024b62a36942083d52caf2f14c0591336d3422
# via
# -r requirements_formatting.txt.in
# pygithub
toml==0.10.2 \
--hash=sha256:806143ae5bfb6a3c6e736a764057db0e6a0e05e338b5630894a5f779cabb4f9b \
--hash=sha256:b3bda1d108d5dd99f4a20d24d9c348e91c4db7ab1b749200bded2f839ccbe68f
# via
# darker
# darkgraylib
typing-extensions==4.14.1 \
--hash=sha256:38b39f4aeeab64884ce9f74c94263ef78f3c22467c8724005483154c26648d36 \
--hash=sha256:d1e1e3b58374dc93031d6eda2420a48ea44a36c2b4766a4fdeb3710755731d76
pynacl==1.5.0
# via pygithub
urllib3==2.6.3 \
--hash=sha256:1b62b6884944a57dbe321509ab94fd4d3b307075e0c2eae991ac71ee15ad38ed \
--hash=sha256:bf272323e553dfb2e87d9bfd225ca7b0f467b919d7bbd355436d3fd37cb0acd4
# via
# -r requirements_formatting.txt.in
# pygithub
# requests
wrapt==1.15.0 \
--hash=sha256:02fce1852f755f44f95af51f69d22e45080102e9d00258053b79367d07af39c0 \
--hash=sha256:077ff0d1f9d9e4ce6476c1a924a3332452c1406e59d90a2cf24aeb29eeac9420 \
--hash=sha256:078e2a1a86544e644a68422f881c48b84fef6d18f8c7a957ffd3f2e0a74a0d4a \
--hash=sha256:0970ddb69bba00670e58955f8019bec4a42d1785db3faa043c33d81de2bf843c \
--hash=sha256:1286eb30261894e4c70d124d44b7fd07825340869945c79d05bda53a40caa079 \
--hash=sha256:21f6d9a0d5b3a207cdf7acf8e58d7d13d463e639f0c7e01d82cdb671e6cb7923 \
--hash=sha256:230ae493696a371f1dbffaad3dafbb742a4d27a0afd2b1aecebe52b740167e7f \
--hash=sha256:26458da5653aa5b3d8dc8b24192f574a58984c749401f98fff994d41d3f08da1 \
--hash=sha256:2cf56d0e237280baed46f0b5316661da892565ff58309d4d2ed7dba763d984b8 \
--hash=sha256:2e51de54d4fb8fb50d6ee8327f9828306a959ae394d3e01a1ba8b2f937747d86 \
--hash=sha256:2fbfbca668dd15b744418265a9607baa970c347eefd0db6a518aaf0cfbd153c0 \
--hash=sha256:38adf7198f8f154502883242f9fe7333ab05a5b02de7d83aa2d88ea621f13364 \
--hash=sha256:3a8564f283394634a7a7054b7983e47dbf39c07712d7b177b37e03f2467a024e \
--hash=sha256:3abbe948c3cbde2689370a262a8d04e32ec2dd4f27103669a45c6929bcdbfe7c \
--hash=sha256:3bbe623731d03b186b3d6b0d6f51865bf598587c38d6f7b0be2e27414f7f214e \
--hash=sha256:40737a081d7497efea35ab9304b829b857f21558acfc7b3272f908d33b0d9d4c \
--hash=sha256:41d07d029dd4157ae27beab04d22b8e261eddfc6ecd64ff7000b10dc8b3a5727 \
--hash=sha256:46ed616d5fb42f98630ed70c3529541408166c22cdfd4540b88d5f21006b0eff \
--hash=sha256:493d389a2b63c88ad56cdc35d0fa5752daac56ca755805b1b0c530f785767d5e \
--hash=sha256:4ff0d20f2e670800d3ed2b220d40984162089a6e2c9646fdb09b85e6f9a8fc29 \
--hash=sha256:54accd4b8bc202966bafafd16e69da9d5640ff92389d33d28555c5fd4f25ccb7 \
--hash=sha256:56374914b132c702aa9aa9959c550004b8847148f95e1b824772d453ac204a72 \
--hash=sha256:578383d740457fa790fdf85e6d346fda1416a40549fe8db08e5e9bd281c6a475 \
--hash=sha256:58d7a75d731e8c63614222bcb21dd992b4ab01a399f1f09dd82af17bbfc2368a \
--hash=sha256:5c5aa28df055697d7c37d2099a7bc09f559d5053c3349b1ad0c39000e611d317 \
--hash=sha256:5fc8e02f5984a55d2c653f5fea93531e9836abbd84342c1d1e17abc4a15084c2 \
--hash=sha256:63424c681923b9f3bfbc5e3205aafe790904053d42ddcc08542181a30a7a51bd \
--hash=sha256:64b1df0f83706b4ef4cfb4fb0e4c2669100fd7ecacfb59e091fad300d4e04640 \
--hash=sha256:74934ebd71950e3db69960a7da29204f89624dde411afbfb3b4858c1409b1e98 \
--hash=sha256:75669d77bb2c071333417617a235324a1618dba66f82a750362eccbe5b61d248 \
--hash=sha256:75760a47c06b5974aa5e01949bf7e66d2af4d08cb8c1d6516af5e39595397f5e \
--hash=sha256:76407ab327158c510f44ded207e2f76b657303e17cb7a572ffe2f5a8a48aa04d \
--hash=sha256:76e9c727a874b4856d11a32fb0b389afc61ce8aaf281ada613713ddeadd1cfec \
--hash=sha256:77d4c1b881076c3ba173484dfa53d3582c1c8ff1f914c6461ab70c8428b796c1 \
--hash=sha256:780c82a41dc493b62fc5884fb1d3a3b81106642c5c5c78d6a0d4cbe96d62ba7e \
--hash=sha256:7dc0713bf81287a00516ef43137273b23ee414fe41a3c14be10dd95ed98a2df9 \
--hash=sha256:7eebcdbe3677e58dd4c0e03b4f2cfa346ed4049687d839adad68cc38bb559c92 \
--hash=sha256:896689fddba4f23ef7c718279e42f8834041a21342d95e56922e1c10c0cc7afb \
--hash=sha256:96177eb5645b1c6985f5c11d03fc2dbda9ad24ec0f3a46dcce91445747e15094 \
--hash=sha256:96e25c8603a155559231c19c0349245eeb4ac0096fe3c1d0be5c47e075bd4f46 \
--hash=sha256:9d37ac69edc5614b90516807de32d08cb8e7b12260a285ee330955604ed9dd29 \
--hash=sha256:9ed6aa0726b9b60911f4aed8ec5b8dd7bf3491476015819f56473ffaef8959bd \
--hash=sha256:a487f72a25904e2b4bbc0817ce7a8de94363bd7e79890510174da9d901c38705 \
--hash=sha256:a4cbb9ff5795cd66f0066bdf5947f170f5d63a9274f99bdbca02fd973adcf2a8 \
--hash=sha256:a74d56552ddbde46c246b5b89199cb3fd182f9c346c784e1a93e4dc3f5ec9975 \
--hash=sha256:a89ce3fd220ff144bd9d54da333ec0de0399b52c9ac3d2ce34b569cf1a5748fb \
--hash=sha256:abd52a09d03adf9c763d706df707c343293d5d106aea53483e0ec8d9e310ad5e \
--hash=sha256:abd8f36c99512755b8456047b7be10372fca271bf1467a1caa88db991e7c421b \
--hash=sha256:af5bd9ccb188f6a5fdda9f1f09d9f4c86cc8a539bd48a0bfdc97723970348418 \
--hash=sha256:b02f21c1e2074943312d03d243ac4388319f2456576b2c6023041c4d57cd7019 \
--hash=sha256:b06fa97478a5f478fb05e1980980a7cdf2712015493b44d0c87606c1513ed5b1 \
--hash=sha256:b0724f05c396b0a4c36a3226c31648385deb6a65d8992644c12a4963c70326ba \
--hash=sha256:b130fe77361d6771ecf5a219d8e0817d61b236b7d8b37cc045172e574ed219e6 \
--hash=sha256:b56d5519e470d3f2fe4aa7585f0632b060d532d0696c5bdfb5e8319e1d0f69a2 \
--hash=sha256:b67b819628e3b748fd3c2192c15fb951f549d0f47c0449af0764d7647302fda3 \
--hash=sha256:ba1711cda2d30634a7e452fc79eabcadaffedf241ff206db2ee93dd2c89a60e7 \
--hash=sha256:bbeccb1aa40ab88cd29e6c7d8585582c99548f55f9b2581dfc5ba68c59a85752 \
--hash=sha256:bd84395aab8e4d36263cd1b9308cd504f6cf713b7d6d3ce25ea55670baec5416 \
--hash=sha256:c99f4309f5145b93eca6e35ac1a988f0dc0a7ccf9ccdcd78d3c0adf57224e62f \
--hash=sha256:ca1cccf838cd28d5a0883b342474c630ac48cac5df0ee6eacc9c7290f76b11c1 \
--hash=sha256:cd525e0e52a5ff16653a3fc9e3dd827981917d34996600bbc34c05d048ca35cc \
--hash=sha256:cdb4f085756c96a3af04e6eca7f08b1345e94b53af8921b25c72f096e704e145 \
--hash=sha256:ce42618f67741d4697684e501ef02f29e758a123aa2d669e2d964ff734ee00ee \
--hash=sha256:d06730c6aed78cee4126234cf2d071e01b44b915e725a6cb439a879ec9754a3a \
--hash=sha256:d5fe3e099cf07d0fb5a1e23d399e5d4d1ca3e6dfcbe5c8570ccff3e9208274f7 \
--hash=sha256:d6bcbfc99f55655c3d93feb7ef3800bd5bbe963a755687cbf1f490a71fb7794b \
--hash=sha256:d787272ed958a05b2c86311d3a4135d3c2aeea4fc655705f074130aa57d71653 \
--hash=sha256:e169e957c33576f47e21864cf3fc9ff47c223a4ebca8960079b8bd36cb014fd0 \
--hash=sha256:e20076a211cd6f9b44a6be58f7eeafa7ab5720eb796975d0c03f05b47d89eb90 \
--hash=sha256:e826aadda3cae59295b95343db8f3d965fb31059da7de01ee8d1c40a60398b29 \
--hash=sha256:eef4d64c650f33347c1f9266fa5ae001440b232ad9b98f1f43dfe7a79435c0a6 \
--hash=sha256:f2e69b3ed24544b0d3dbe2c5c0ba5153ce50dcebb576fdc4696d52aa22db6034 \
--hash=sha256:f87ec75864c37c4c6cb908d282e1969e79763e0d9becdfe9fe5473b7bb1e5f09 \
--hash=sha256:fbec11614dba0424ca72f4e8ba3c420dba07b4a7c206c8c8e4e73f2e98f4c559 \
--hash=sha256:fd69666217b62fa5d7c6aa88e507493a34dec4fa20c5bd925e4bc12fce586639
requests==2.31.0
# via pygithub
toml==0.10.2
# via darker
urllib3==2.0.4
# via requests
wrapt==1.15.0
# via deprecated
@@ -1,9 +0,0 @@
black~=25.1
darker==2.1.1
PyGithub==2.6.1
cryptography>=46.0.5
urllib3>=2.6.3
requests>=2.32.4
idna>=3.7
certifi>=2024.7.4
PyNaCl>=1.6.2
+1 -1
Vendored Submodule
+1
Submodule External/jemalloc added at 02ca52b5fe.
Submodule External/range-v3 deleted from ca1388fb9d.
Vendored Submodule
+1
Submodule External/robin-map added at d5683d9f18.
Submodule External/rpmalloc deleted from 1f6fb494f2.
-3
View File
@@ -1,6 +1,3 @@
set(NAME tiny-json)
set(SRCS tiny-json.c)
add_library(${NAME} STATIC ${SRCS})
target_include_directories(${NAME} PUBLIC ${CMAKE_CURRENT_LIST_DIR})
add_library(${NAME}::${NAME} ALIAS ${NAME})
+1 -1
+1 -1
-1
Submodule External/zydis deleted from 9bfadd6a55.
+43 -10
View File
@@ -1,16 +1,16 @@
cmake_minimum_required(VERSION 3.14)
set(PROJECT_NAME FEXCore)
set (PROJECT_NAME FEXCore)
project(${PROJECT_NAME}
VERSION 0.01
LANGUAGES CXX)
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
set(ARCHITECTURE_x86_64 1)
set(_M_X86_64 1)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
set(ARCHITECTURE_arm64 1)
set(_M_ARM_64 1)
endif()
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
@@ -24,10 +24,45 @@ include(CheckCXXCompilerFlag)
include(CheckIncludeFileCXX)
include(CheckCXXSourceCompiles)
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
# Useful to have for freestanding libFEXCore
add_subdirectory(External/vixl/)
include_directories(External/vixl/src/)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
configure_file(${CMAKE_CURRENT_SOURCE_DIR}/include/git_version.h.in
set(GIT_SHORT_HASH "Unknown")
set(GIT_DESCRIBE_STRING "FEX-Unknown")
if (OVERRIDE_VERSION STREQUAL "detect")
# Find our git hash
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} rev-parse --short=7 HEAD
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_SHORT_HASH
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
endif()
else()
set(GIT_SHORT_HASH "${OVERRIDE_VERSION}")
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
endif()
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/include/git_version.h.in
${CMAKE_BINARY_DIR}/generated/git_version.h)
include_directories(${CMAKE_BINARY_DIR}/generated)
@@ -39,12 +74,10 @@ add_compile_options($<$<COMPILE_LANGUAGE:CXX>:-fno-strict-aliasing> $<$<COMPILE_
add_subdirectory(Source/)
if (NOT BUILD_STEAM_SUPPORT)
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
DESTINATION include
COMPONENT Development)
endif()
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
DESTINATION include
COMPONENT Development)
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
+174 -24
View File
@@ -118,6 +118,41 @@ def print_man_env_option(name, desc, default, no_json_key):
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
def print_man_options(options):
output_man.write(".Sh OPTIONS\n")
output_man.write(".Bl -tag -width -indent\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
default = op_vals["Default"]
value_type = op_vals["Type"]
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = op_vals["TextDefault"]
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
# Wrap the string argument in quotes
default = "'" + default + "'"
print_man_option(
short,
long,
op_vals["Desc"],
default
)
if (value_type == "strenum"):
Enums = op_vals["Enums"]
output_man.write("\\fBAvailable Options:\\fR\n")
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
output_man.write("\n.sp\n")
output_man.write(".El\n")
def print_man_environment(options):
output_man.write(".Sh ENVIRONMENT\n")
output_man.write(".Bl -tag -width -indent\n")
@@ -156,10 +191,10 @@ def print_man_environment_tail():
"APP_CONFIG_LOCATION",
[
"Allows the user to override where FEX looks for configuration files",
"By default FEX will look in ${XDG_CONFIG_HOME, $HOME/.config}/fex-emu/",
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
"This will override the full path",
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -168,12 +203,12 @@ def print_man_environment_tail():
"APP_CONFIG",
[
"Allows the user to override where FEX looks for only the application config file",
"By default FEX will look in ${XDG_CONFIG_HOME, $HOME/.config}/fex-emu/Config.json",
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/Config.json",
"This will override this file location",
"One must be careful with this option as it will override any applications that load with execve as well"
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -182,7 +217,7 @@ def print_man_environment_tail():
"APP_DATA_LOCATION",
[
"Allows the user to override where FEX looks for data files",
"By default FEX will look in {$XDG_DATA_HOME, $HOME/.local/share}/fex-emu/",
"By default FEX will look in {$HOME, $XDG_DATA_HOME}/.fex-emu/",
"This will override the full path",
"This is the folder where FEX stores generated files like IR cache"
],
@@ -192,34 +227,33 @@ def print_man_environment_tail():
"PORTABLE",
[
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
"For FEX on Linux:",
"These files are instead read from <FEXPath>/fex-emu/ by default.",
"For FEXInterpreter on Linux:",
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
"For Arm64ec/Wow64 WINE builds:",
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
],
"''", True)
print_man_env_option(
"APP_CACHE_LOCATION",
[
"Allows the user to override where FEX stores and loads cache files",
"By default FEX will look in ${XDG_CACHE_HOME, $HOME/.cache}/fex-emu/",
"This will override the full path, trailing forward-slash is expected to exist",
],
"''", True)
def print_man_header():
header ='''.Dd {0}
.Dt FEX
.Os Linux
.Sh NAME
.Nm FEX
.Nm FEXLoader
.Nm FEXInterpreter
.Nm FEXBash
.Nd Fast x86-64 and x86 emulation.
.Sh SYNOPSIS
.Nm
.Ar <args> ...
.Op options
.Op Ar --
.Ar Application
<args> ...
.Pp
.Nm FEXInterpreter
.Ar Application
<args> ...
.Pp
.Nm FEXBash
.Ar <args> ...
@@ -234,7 +268,7 @@ FEX is very much work in progress, so expect things to change.
def print_man_tail():
tail ='''.Sh FILES
.Bl -tag -width "$prefix/share/fex-emu/GuestThunks" -compact
.It Pa $XDG_CONFIG_DIR/fex-emu
.It Pa $XDG_HOME_DIR/.fex-emu
Default FEX user configuration directory
.It Pa $prefix/share/fex-emu/AppConfig
System level application configuration files
@@ -327,6 +361,82 @@ def print_config_option(type, group_name, json_name, default_value, short, choic
output_argloader.write("\n");
def print_argloader_options(options):
output_argloader.write("#ifdef BEFORE_PARSE\n")
output_argloader.write("#undef BEFORE_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
default = op_vals["Default"]
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
# Wrap the string argument in quotes
default = "\"" + default + "\""
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = "\"" + op_vals["TextDefault"] + "\""
short = None
choices = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if ("Choices" in op_vals):
choices = op_vals["Choices"]
print_config_option(
op_vals["Type"],
op_group,
op_key,
default,
short,
choices,
op_vals["Desc"])
output_argloader.write("\n")
output_argloader.write("#endif\n")
def print_parse_argloader_options(options):
output_argloader.write("#ifdef AFTER_PARSE\n")
output_argloader.write("#undef AFTER_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
output_argloader.write("if (Options.is_set_by_user(\"{0}\")) {{\n".format(op_key))
value_type = op_vals["Type"]
NeedsString = False
conversion_func = "fextl::fmt::format(\"{}\", "
if ("ArgumentHandler" in op_vals):
NeedsString = True
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
if (value_type == "str"):
NeedsString = True
conversion_func = "std::move("
if (value_type == "bool"):
# boolean values need a decimal specifier. Otherwise fmt prints strings.
conversion_func = "fextl::fmt::format(\"{:d}\", "
if (value_type == "strenum"):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key, op_key))
elif (value_type == "strarray"):
# these need a bit more help
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
output_argloader.write("\t}\n")
else:
if (NeedsString):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
else:
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_envloader_options(options):
output_argloader.write("#ifdef ENVLOADER\n")
output_argloader.write("#undef ENVLOADER\n")
@@ -337,13 +447,13 @@ def print_parse_envloader_options(options):
value_type = op_vals["Type"]
if (value_type == "strenum"):
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
output_argloader.write("\tValue = FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View);\n".format(op_key, op_key))
output_argloader.write("Value = FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View);\n".format(op_key, op_key, op_key))
output_argloader.write("}\n")
if ("ArgumentHandler" in op_vals):
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
output_argloader.write("\tValue = {0}(Value_View);\n".format(conversion_func))
output_argloader.write("Value = {0}(Value_View);\n".format(conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
@@ -357,15 +467,15 @@ def print_parse_jsonloader_options(options):
value_type = op_vals["Type"]
if (value_type == "strenum"):
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key))
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
output_argloader.write("}\n")
elif (value_type == "strarray"):
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
output_argloader.write("\tAppendStrArrayValue(KeyOption, ConfigString);\n")
output_argloader.write("}\n")
assert op_key is not None, "No options found in JSONLOADER"
output_argloader.write("else {\n")
output_argloader.write("\tSet(KeyOption, ConfigString);\n")
output_argloader.write("else {{\n".format(op_key))
output_argloader.write("Set(KeyOption, ConfigString);\n")
output_argloader.write("}\n")
output_argloader.write("#endif\n")
@@ -407,6 +517,41 @@ def print_parse_enum_options(options):
output_argloader.write("#endif\n")
def check_for_duplicate_options(options):
short_map = []
long_map = []
# Spin through all the items and see if we have a duplicate option
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
long_invert = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if (op_vals["Type"] == "bool"):
long_invert = "no-" + long
# Check for short key duplication
if (short != None):
if (short in short_map):
raise Exception("Short config '{0}' for option '{1}' has duplicate entry!".format(short, op_key))
else:
short_map.append(short)
# Check for long key duplication
if (long in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long))
else:
long_map.append(long)
# Check for long key duplication
if (long_invert != None):
if (long_invert in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long_invert))
else:
long_map.append(long_invert)
if (len(sys.argv) < 5):
sys.exit()
@@ -423,6 +568,8 @@ json_object = json.loads(json_text)
options = json_object["Options"]
unnamed_options = json_object["UnnamedOptions"]
check_for_duplicate_options(options)
# Generate config include file
output_file = open(output_filename, "w")
print_header()
@@ -434,6 +581,7 @@ output_file.close()
# Generate man file
output_man = open(output_man_page, "w")
print_man_header()
print_man_options(options)
print_man_environment(options)
print_man_tail()
@@ -441,6 +589,8 @@ output_man.close()
# Generate argument loader code
output_argloader = open(output_argumentloader_filename, "w")
print_argloader_options(options);
print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
+72 -95
View File
@@ -58,10 +58,9 @@ class OpDefinition:
JITDispatch: bool
JITDispatchOverride: str
TiedSource: int
Inline: list[str]
Arguments: list[OpArgument]
EmitValidation: list[str]
Desc: list[str]
Arguments: list
EmitValidation: list
Desc: list
def __init__(self):
self.Name = None
@@ -92,14 +91,19 @@ class OpDefinition:
attrs = vars(self)
print(", ".join("%s: %s" % item for item in attrs.items()))
IRTypesToCXX: dict[str, IRType] = {}
CXXTypeToIR: dict[str, IRType] = {}
IROps: list[OpDefinition] = []
IRTypesToCXX = {}
CXXTypeToIR = {}
IROps = []
IROpNameSet: set[str] = set()
IROpNameMap = {}
def is_ssa_type(op_type: str):
return op_type in {"SSA", "GPR", "GPRPair", "FPR"}
def is_ssa_type(type):
if (type == "SSA" or
type == "GPR" or
type == "GPRPair" or
type == "FPR"):
return True
return False
def parse_irtypes(irtypes):
for op_key, op_val in irtypes.items():
@@ -214,8 +218,11 @@ def parse_ops(ops):
OpArg.DefaultInitializer = DefaultInit[1][:-1]
# If SSA type then we can generate validation for this op
if OpArg.IsSSA and OpArg.Type in {"GPR", "GPRPair", "FPR"}:
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == RegClass::Invalid || WalkFindRegClass({ArgName}) == RegClass::{OpArg.Type}")
if (OpArg.IsSSA and
(OpArg.Type == "GPR" or
OpArg.Type == "GPRPair" or
OpArg.Type == "FPR")):
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
OpArg.Name = ArgName
OpArg.NameWithPrefix = NameWithPrefix
@@ -271,12 +278,6 @@ def parse_ops(ops):
if "TiedSource" in op_val:
OpDef.TiedSource = op_val["TiedSource"]
# Pad Inline out to the argument count
OpDef.Inline = [''] * len(OpDef.Arguments)
if "Inline" in op_val:
Value = op_val["Inline"]
OpDef.Inline[0:len(Value)] = Value
# Do some fixups of the data here
if len(OpDef.EmitValidation) != 0:
for i in range(len(OpDef.EmitValidation)):
@@ -288,28 +289,21 @@ def parse_ops(ops):
#OpDef.print()
# Error on duplicate op
if OpDef.Name in IROpNameSet:
if OpDef.Name in IROpNameMap:
ExitError("Duplicate Op defined! {}".format(OpDef.Name))
IROps.append(OpDef)
IROpNameSet.add(OpDef.Name)
IROpNameMap[OpDef.Name] = 1
# Print out enum values
def print_enums(enums):
def print_enums():
output_file.write("#ifdef IROP_ENUM\n")
output_file.write("enum IROps : uint16_t {\n")
for op in IROps:
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
output_file.write("};\n")
for name, members in enums.items():
output_file.write(f"enum {name} {{\n")
for member in members:
if member:
output_file.write(f"\t{member}\n")
else:
output_file.write("\n")
output_file.write("};\n\n")
output_file.write("};\n")
output_file.write("#undef IROP_ENUM\n")
output_file.write("#endif\n\n")
@@ -403,16 +397,16 @@ def print_ir_sizes():
// Make sure our array maps directly to the IROps enum
static_assert(IRSizes[IROps::OP_LAST] == -1ULL);
[[nodiscard]] inline size_t GetSize(IROps Op) { return IRSizes[Op]; }
[[nodiscard, gnu::const]] std::string_view const& GetName(IROps Op);
[[nodiscard, gnu::const]] uint8_t GetArgs(IROps Op);
[[nodiscard, gnu::const]] uint8_t GetRAArgs(IROps Op);
[[nodiscard, gnu::const]] FEXCore::IR::RegClass GetRegClass(IROps Op);
[[nodiscard, gnu::const]] bool HasSideEffects(IROps Op);
[[nodiscard, gnu::const]] bool ImplicitFlagClobber(IROps Op);
[[nodiscard, gnu::const]] bool GetHasDest(IROps Op);
[[nodiscard, gnu::const]] bool LoweredX87(IROps Op);
[[nodiscard, gnu::const]] int8_t TiedSource(IROps Op);
[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }
[[nodiscard, gnu::const, gnu::visibility("default")]] std::string_view const& GetName(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetArgs(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetRAArgs(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] bool HasSideEffects(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] bool ImplicitFlagClobber(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] bool GetHasDest(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] bool LoweredX87(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] int8_t TiedSource(IROps Op);
#undef IROP_SIZES
#endif
@@ -421,29 +415,30 @@ def print_ir_sizes():
def print_ir_reg_classes():
output_file.write("#ifdef IROP_REG_CLASSES_IMPL\n")
output_file.write("constexpr std::array<FEXCore::IR::RegClass, IROps::OP_LAST + 1> IRRegClasses = {\n")
output_file.write("constexpr std::array<FEXCore::IR::RegisterClassType, IROps::OP_LAST + 1> IRRegClasses = {\n")
for op in IROps:
if op.Name == "Last":
output_file.write("\tRegClass::Invalid,\n")
output_file.write("\tFEXCore::IR::InvalidClass,\n")
else:
if op.HasDest and op.DestType is None:
Class = "Invalid"
if op.HasDest and op.DestType == None:
ExitError("IR op {} has destination with no destination class".format(op.Name))
if op.HasDest and op.DestType == "SSA": # Special case SSA type
output_file.write("\tRegClass::Complex,\n")
output_file.write("\tFEXCore::IR::ComplexClass,\n")
elif op.HasDest:
output_file.write("\tRegClass::{},\n".format(op.DestType))
output_file.write("\tFEXCore::IR::{}Class,\n".format(op.DestType))
else:
# No destination so it has an invalid destination class
output_file.write("\tRegClass::Invalid, // No destination\n")
output_file.write("\tFEXCore::IR::InvalidClass, // No destination\n")
output_file.write("};\n\n")
output_file.write("// Make sure our array maps directly to the IROps enum\n")
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == RegClass::Invalid);\n\n")
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == FEXCore::IR::InvalidClass);\n\n")
output_file.write("FEXCore::IR::RegClass GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
output_file.write("FEXCore::IR::RegisterClassType GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
output_file.write("#undef IROP_REG_CLASSES_IMPL\n")
output_file.write("#endif\n\n")
@@ -566,7 +561,9 @@ def print_ir_arg_printer():
SSAArgNum = 0
FirstArg = True
for arg in op.Arguments:
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
# No point printing temporaries that we can't recover
if arg.Temporary:
continue
@@ -591,13 +588,13 @@ def print_ir_arg_printer():
output_file.write("#endif\n")
def print_validation(op):
if len(op.EmitValidation) != 0:
output_file.write("#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
if op.EmitValidation != None:
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
for Validation in op.EmitValidation:
Sanitized = Validation.replace("\"", "\\\"")
output_file.write("\t\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
output_file.write("#endif\n")
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
output_file.write("\t\t#endif\n")
# Print out IR allocator helpers
def print_ir_allocator_helpers():
@@ -667,7 +664,7 @@ def print_ir_allocator_helpers():
output_file.write("\t\treturn HeaderOp->Op;\n")
output_file.write("\t}\n\n")
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n")
output_file.write("\tFEXCore::IR::RegisterClassType GetOpRegClass(const OrderedNode *Op) const {\n")
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
output_file.write("\t}\n\n")
@@ -678,25 +675,25 @@ def print_ir_allocator_helpers():
# Generate helpers with operands
for op in IROps:
if op.Name != "Last":
output_file.write("\t///\n".join(["\t/// {}\n" .format(comment) for comment in op.Desc]))
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
# Output SSA args first
for i, arg in enumerate(op.Arguments):
LastArg = i == len(op.Arguments) - 1
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
if arg.Temporary:
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
elif arg.IsSSA:
# SSA value
output_file.write("OrderedNodeWrapper {}".format(arg.Name))
else:
# User defined op that is stored
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
if arg.DefaultInitializer:
if arg.DefaultInitializer != None:
output_file.write(" = {}".format(arg.DefaultInitializer))
if not LastArg:
@@ -752,22 +749,22 @@ def print_ir_allocator_helpers():
# Now do the OrderedNode * version if necessary
if op.SSAArgNum:
output_file.write("\t///\n".join(["\t/// {}\n" .format(comment) for comment in op.Desc]))
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
for i, arg in enumerate(op.Arguments):
LastArg = i == len(op.Arguments) - 1
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
if arg.Temporary:
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
elif arg.IsSSA:
output_file.write("OrderedNode *{}".format(arg.Name))
else:
CType = IRTypesToCXX[arg.Type].CXXName
output_file.write("{} {}".format(CType, arg.Name))
output_file.write("{} {}".format(CType, arg.Name));
if arg.DefaultInitializer:
if arg.DefaultInitializer != None:
output_file.write(" = {}".format(arg.DefaultInitializer))
if not LastArg:
@@ -776,29 +773,9 @@ def print_ir_allocator_helpers():
output_file.write(") {\n")
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
idx = 0
for arg in op.Arguments:
if arg.IsSSA:
# Inline an immediate if we can
inline = op.Inline[idx]
idx += 1
if inline != '':
Sized = "Size" in [x.Name for x in op.Arguments]
P = ["Size" if Sized else "OpSize::i64Bit", arg.Name]
# A few cases need extra info plumbed.
if inline == "SubtractZero":
P += ["Src2"]
elif inline == "Mem":
P += ["OffsetType", "OffsetScale"]
elif inline == "Memtso":
P += ["OffsetType", "OffsetScale", "true /* TSO */"]
inline = "Mem"
output_file.write(f"\t\t{arg.Name} = Inline{inline}({', '.join(P)});\n")
output_file.write(f"\t\t{arg.Name}->AddUse();\n")
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
# Insert validation here. This is skipped for the
# OrderedNodeWrapper version because validation can depend on
@@ -808,15 +785,16 @@ def print_ir_allocator_helpers():
print_validation(op)
output_file.write(f"\t\treturn _{op.Name}(")
for i, arg in enumerate(op.Arguments):
LastArg = i == len(op.Arguments) - 1
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
output_file.write(arg.Name)
if arg.IsSSA:
output_file.write("->Wrapped(ListDataBegin)")
if not LastArg:
output_file.write(", ")
output_file.write(");\n")
output_file.write("\t}\n\n")
output_file.write(");\n");
output_file.write("\t}\n\n");
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
output_file.write("#endif\n")
@@ -847,8 +825,8 @@ def print_ir_dispatcher_dispatch():
output_dispatch_file.write("#endif\n")
if len(sys.argv) < 4:
ExitError("Insufficient parameters passed to script")
if (len(sys.argv) < 4):
ExitError()
output_filename = sys.argv[2]
output_dispatcher_filename = sys.argv[3]
@@ -860,7 +838,6 @@ json_file.close()
json_object = json.loads(json_text)
json_object = {k.upper(): v for k, v in json_object.items()}
enums = json_object["ENUMS"]
ops = json_object["OPS"]
irtypes = json_object["IRTYPES"]
defines = json_object["DEFINES"]
@@ -870,7 +847,7 @@ parse_ops(ops)
output_file = open(output_filename, "w")
print_enums(enums)
print_enums()
print_ir_structs(defines)
print_ir_sizes()
print_ir_reg_classes()
+84 -70
View File
@@ -1,28 +1,32 @@
set(MAN_DIR share/man CACHE PATH "MAN_DIR")
include(GNUInstallDirs)
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
set(FEXCORE_BASE_SRCS
set (FEXCORE_BASE_SRCS
Interface/Config/Config.cpp
Utils/Allocator.cpp
Utils/FileLoading.cpp
Utils/ForcedAssert.cpp
Utils/LogManager.cpp
Utils/SpinWaitLock.cpp)
Utils/SpinWaitLock.cpp
)
if (NOT MINGW)
if (NOT MINGW_BUILD)
list(APPEND FEXCORE_BASE_SRCS
Utils/Allocator/64BitAllocator.cpp)
endif()
set(SRCS
set (SRCS
Common/JitSymbols.cpp
Interface/Context/Context.cpp
Interface/Core/LookupCache.cpp
Interface/Core/CodeCache.cpp
Interface/Core/Core.cpp
Interface/Core/CPUBackend.cpp
Interface/Core/Addressing.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/ObjectCache/JobHandling.cpp
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
Interface/Core/ObjectCache/ObjectCacheService.cpp
Interface/Core/OpcodeDispatcher/AVX_128.cpp
Interface/Core/OpcodeDispatcher/Crypto.cpp
Interface/Core/OpcodeDispatcher/Flags.cpp
@@ -30,6 +34,8 @@ set(SRCS
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher/X87F64.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/X86Tables.cpp
Interface/Core/X86HelperGen.cpp
Interface/Core/ArchHelpers/Arm64Emitter.cpp
Interface/Core/Dispatcher/Dispatcher.cpp
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
@@ -56,20 +62,22 @@ set(SRCS
Interface/Core/X86Tables/VEXTables.cpp
Interface/Core/X86Tables/X87Tables.cpp
Interface/GDBJIT/GDBJIT.cpp
Interface/IR/AOTIR.cpp
Interface/IR/IRDumper.cpp
Interface/IR/IREmitter.cpp
Interface/IR/PassManager.cpp
Interface/IR/Passes/ConstProp.cpp
Interface/IR/Passes/IRDumperPass.cpp
Interface/IR/Passes/IRValidation.cpp
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/x87StackOptimizationPass.cpp
Utils/LongJump.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/Profiler.cpp)
Utils/Profiler.cpp
)
if (ARCHITECTURE_arm64)
if (_M_ARM_64)
list(APPEND SRCS Utils/ArchHelpers/Arm64.cpp)
else()
list(APPEND SRCS Utils/ArchHelpers/Arm64_stubs.cpp)
@@ -82,49 +90,42 @@ endif()
set(DEFINES -DJIT_ARM64)
if (ARCHITECTURE_x86_64)
list(APPEND DEFINES -DARCHITECTURE_x86_64=1)
if (_M_X86_64)
list(APPEND DEFINES -D_M_X86_64=1)
endif()
if (ARCHITECTURE_arm64)
list(APPEND DEFINES -DARCHITECTURE_arm64=1)
if (_M_ARM_64)
list(APPEND DEFINES -D_M_ARM_64=1)
endif()
if (ENABLE_VIXL_DISASSEMBLER)
list(APPEND DEFINES -DVIXL_DISASSEMBLER=1)
endif()
if (ENABLE_ZYDIS)
list(APPEND DEFINES -DZYDIS_DISASSEMBLER=1)
endif()
if (ARCHITECTURE_arm64 AND HAS_CLANG_PRESERVE_ALL)
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=__attribute__((preserve_all));-DFEXCORE_HAS_PRESERVE_ALL_ATTR=1")
else()
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=;-DFEXCORE_HAS_PRESERVE_ALL_ATTR=0")
endif()
set(LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter cephes_128bit)
set (LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter cephes_128bit)
if (ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
list(APPEND LIBS vixl::vixl)
list (APPEND LIBS vixl)
endif()
if (ENABLE_ZYDIS)
list(APPEND LIBS Zydis::Zydis)
endif()
if (NOT MINGW)
list(APPEND LIBS dl)
if (NOT MINGW_BUILD)
list (APPEND LIBS dl)
else()
list(APPEND LIBS synchronization)
if (ARCHITECTURE_arm64ec)
list(APPEND LIBS mincore)
list (APPEND LIBS synchronization)
if (_M_ARM_64EC)
list (APPEND LIBS mincore)
endif()
endif()
# Generate config
configure_file(${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
${CMAKE_BINARY_DIR}/generated/Config/Config.json)
# Generate IR include file
@@ -139,10 +140,11 @@ add_custom_command(
OUTPUT "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
DEPENDS "${INPUT_NAME}"
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
"${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}")
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
)
set_source_files_properties(${OUTPUT_NAME} PROPERTIES GENERATED TRUE)
set_source_files_properties(${OUTPUT_NAME} PROPERTIES
GENERATED TRUE)
# Generate IR documentation
set(OUTPUT_IR_DOC "${CMAKE_BINARY_DIR}/IR.md")
@@ -151,10 +153,11 @@ add_custom_command(
OUTPUT "${OUTPUT_IR_DOC}"
DEPENDS "${INPUT_NAME}"
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
"${INPUT_NAME}" "${OUTPUT_IR_DOC}")
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py" "${INPUT_NAME}" "${OUTPUT_IR_DOC}"
)
set_source_files_properties(${OUTPUT_IR_NAME} PROPERTIES GENERATED TRUE)
set_source_files_properties(${OUTPUT_IR_NAME} PROPERTIES
GENERATED TRUE)
# Create the target
add_custom_target(IR_INC
@@ -178,12 +181,14 @@ add_custom_command(
DEPENDS "${INPUT_CONFIG_NAME}"
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py"
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py" "${INPUT_CONFIG_NAME}" "${OUTPUT_CONFIG_NAME}" "${OUTPUT_MAN_NAME}"
"${OUTPUT_CONFIG_OPTION_NAME}")
"${OUTPUT_CONFIG_OPTION_NAME}"
)
add_custom_command(
OUTPUT "${OUTPUT_MAN_NAME_COMPRESS}"
DEPENDS "${OUTPUT_MAN_NAME}"
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}")
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}"
)
set_source_files_properties(${OUTPUT_CONFIG_NAME} PROPERTIES
GENERATED TRUE)
@@ -202,10 +207,8 @@ add_custom_target(CONFIG_INC
DEPENDS "${OUTPUT_MAN_NAME}"
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
if (NOT BUILD_STEAM_SUPPORT)
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
endif()
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
# Add in diagnostic colours if the option is available.
# Ninja code generator will kill colours if this isn't here
@@ -227,7 +230,8 @@ function(AddDefaultOptionsToTarget Name)
target_compile_definitions(${Name} PRIVATE ${DEFINES})
add_dependencies(${Name} CONFIG_INC IR_INC)
target_compile_options(${Name} PRIVATE
target_compile_options(${Name}
PRIVATE
-Wall
-Werror=cast-qual
-Werror=ignored-qualifiers
@@ -235,72 +239,82 @@ function(AddDefaultOptionsToTarget Name)
-Wno-trigraphs
-ffunction-sections
-fwrapv)
-fwrapv
)
if (GCC_COLOR)
target_compile_options(${Name} PRIVATE "-fdiagnostics-color=always")
target_compile_options(${Name}
PRIVATE
"-fdiagnostics-color=always")
endif()
if (CLANG_COLOR)
target_compile_options(${Name} PRIVATE "-fcolor-diagnostics")
target_compile_options(${Name}
PRIVATE
"-fcolor-diagnostics")
endif()
LinkerGC(${Name})
target_link_libraries(${Name} PUBLIC unordered_dense::unordered_dense)
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
target_link_options(${Name}
PRIVATE
"LINKER:--gc-sections"
"LINKER:--strip-all"
"LINKER:--as-needed"
)
endif()
endfunction()
# Build FEXCore_Base static library
# Build FEXCore_Config static library
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
target_link_libraries(FEXCore_Base PUBLIC ${LIBS})
target_link_libraries(FEXCore_Base ${LIBS})
AddDefaultOptionsToTarget(FEXCore_Base)
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
target_link_libraries(FEXCore_Base PUBLIC TracyClient)
target_link_libraries(FEXCore_Base TracyClient)
endif()
function(AddObject Name)
add_library(${Name} OBJECT ${SRCS})
function(AddObject Name Type)
add_library(${Name} ${Type} ${SRCS})
target_link_libraries(${Name} PRIVATE FEXCore_Base)
target_link_libraries(${Name} FEXCore_Base)
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
AddDefaultOptionsToTarget(${Name})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
endfunction()
function(AddLibrary Name Type)
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
target_link_libraries(${Name} FEXCore_Base)
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
# During generation of the import library (dll.a), MinGW needs some extra symbols from libraries
# such as fmt, which are propagated by FEXCore_Base. Wonderful.
if (MINGW)
target_link_libraries(${Name} PRIVATE FEXCore_Base)
endif()
AddDefaultOptionsToTarget(${Name})
endfunction()
AddObject(${PROJECT_NAME}_object)
AddObject(${PROJECT_NAME}_object OBJECT)
AddLibrary(${PROJECT_NAME} STATIC)
AddLibrary(${PROJECT_NAME}_shared SHARED)
if (NOT MINGW AND NOT BUILD_STEAM_SUPPORT)
install(TARGETS ${PROJECT_NAME}_shared LIBRARY
DESTINATION ${CMAKE_INSTALL_LIBDIR}
COMPONENT Libraries)
if (NOT MINGW_BUILD)
install(TARGETS ${PROJECT_NAME}_shared
LIBRARY
DESTINATION ${CMAKE_INSTALL_LIBDIR}
COMPONENT Libraries)
endif()
# Meta-library to link jemalloc libraries enabled in the build configuration.
# Only needed for targets that run emulation. For others, use JemallocDummy.
add_library(JemallocLibs STATIC Utils/AllocatorHooks.cpp)
if (ENABLE_FEX_ALLOCATOR)
target_compile_definitions(JemallocLibs PRIVATE ENABLE_FEX_ALLOCATOR=1)
target_link_libraries(JemallocLibs PUBLIC rpmalloc)
if (ENABLE_JEMALLOC)
target_compile_definitions(JemallocLibs PRIVATE ENABLE_JEMALLOC=1 JEMALLOC_NO_RENAME=1)
target_link_libraries(JemallocLibs PUBLIC FEX_jemalloc)
endif()
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
set_source_files_properties(Interface/HLE/Thunks/Thunks.cpp PROPERTIES COMPILE_DEFINITIONS ENABLE_JEMALLOC_GLIBC=1)
target_link_libraries(JemallocLibs INTERFACE FEX_jemalloc_glibc)
endif()
if (NOT MINGW)
if (NOT MINGW_BUILD)
# Dummy project to use for host tools.
# This overrides use of jemalloc in FEXCore with the normal glibc allocator.
add_library(JemallocDummy STATIC Utils/AllocatorHooks.cpp)
@@ -308,4 +322,4 @@ if (NOT MINGW)
endif()
# The shared library should always link enabled jemalloc libraries
target_link_libraries(${PROJECT_NAME}_shared PRIVATE JemallocLibs)
target_link_libraries(${PROJECT_NAME}_shared JemallocLibs)
+5 -4
View File
@@ -1,19 +1,20 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <chrono>
#include <cstddef>
#include <cstdint>
#include <cstdio>
#include <memory>
#include <string_view>
namespace FEXCore {
// Buffered JIT symbol tracking.
struct JITSymbolBuffer {
// Maximum buffer size to ensure we are a page in size.
constexpr static size_t BUFFER_SIZE = FEXCore::Utils::FEX_PAGE_SIZE - (8 * 2);
constexpr static size_t BUFFER_SIZE = 4096 - (8 * 2);
// Maximum distance until the end of the buffer to do a write.
constexpr static size_t NEEDS_WRITE_DISTANCE = BUFFER_SIZE - 64;
// Maximum time threshhold to wait before a buffer write occurs.
@@ -28,7 +29,7 @@ struct JITSymbolBuffer {
size_t Offset {};
char Buffer[BUFFER_SIZE] {};
};
static_assert(sizeof(JITSymbolBuffer) == FEXCore::Utils::FEX_PAGE_SIZE, "Ensure this is one page in size");
static_assert(sizeof(JITSymbolBuffer) == 4096, "Ensure this is one page in size");
class JITSymbols final {
public:
+36 -89
View File
@@ -4,9 +4,9 @@
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/BitUtils.h>
#include "cephes_128bit.h"
#include <bit>
#include <cmath>
#include <cstring>
#include <stdint.h>
@@ -19,7 +19,7 @@ extern "C" {
}
struct FEX_PACKED X80SoftFloat {
#ifdef ARCHITECTURE_x86_64
#ifdef _M_X86_64
// Define this to push some operations to x87
// Only useful to see if precision loss is killing something
// #define DEBUG_X86_FLOAT
@@ -30,33 +30,29 @@ struct FEX_PACKED X80SoftFloat {
#define BIGFLOAT float128_t
#define BIGFLOATSIZE 16
#endif
#elif defined(ARCHITECTURE_arm64)
#elif defined(_M_ARM_64)
#define BIGFLOAT float128_t
#define BIGFLOATSIZE 16
#else
#error No 128bit float for this target!
#endif
uint64_t Significand;
union {
uint16_t Raw;
struct {
uint16_t Exponent : 15;
uint16_t Sign : 1;
};
} Top;
uint64_t Significand : 64;
uint16_t Exponent : 15;
uint16_t Sign : 1;
X80SoftFloat() {
memset(this, 0, sizeof(*this));
}
X80SoftFloat(uint16_t _Sign, uint16_t _Exponent, uint64_t _Significand)
: Significand {_Significand}
, Top {.Raw = static_cast<uint16_t>((_Exponent & 0x7FFF) | (_Sign << 15))} {}
, Exponent {_Exponent}
, Sign {_Sign} {}
fextl::string str() const {
fextl::ostringstream string;
string << std::hex << Top.Sign;
string << "_" << Top.Exponent;
string << std::hex << Sign;
string << "_" << Exponent;
string << "_" << (Significand >> 63);
string << "_" << (Significand & ((1ULL << 63) - 1));
return string.str();
@@ -161,30 +157,7 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
/*
* Check for invalid operation cases first - Intel FPREM sets Invalid Operation
* for several cases including infinity dividend and zero divisor.
*/
X80SoftFloat result = 0;
if (HandleInfinityOp(state, lhs, result)) {
return result;
} else if (lhs.Top.Exponent == 0x7FFF && (lhs.Significand & 0x7FFFFFFFFFFFFFFFULL)) { // NaN
// propagate NaN
state->exceptionFlags |= softfloat_flag_invalid;
return lhs;
}
// Check for zero divisor - fprem(x, 0) is invalid operation
if (rhs.Top.Exponent == 0 && rhs.Significand == 0) {
state->exceptionFlags |= softfloat_flag_invalid;
// Return QNaN
result.Top.Sign = 0;
result.Top.Exponent = 0x7FFF;
result.Significand = 0xC000000000000000ULL;
return result;
}
/*
* FPREM is not an IEEE-754 remainder. From the Intel spec:
* FPREM is not an IEEE-754 remainder. From the spec:
*
* Computes the remainder obtained from dividing the value in the ST(0)
* register (the dividend) by the value in the ST(1) register (the divisor
@@ -257,12 +230,12 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
// Zero is a special case, the significand for +/- 0 is +/- zero.
if (lhs.Top.Exponent == 0x0 && lhs.Significand == 0x0) {
if (lhs.Exponent == 0x0 && lhs.Significand == 0x0) {
return lhs;
}
X80SoftFloat Tmp = lhs;
Tmp.Top.Exponent = 0x3FFF;
Tmp.Top.Sign = lhs.Top.Sign;
Tmp.Exponent = 0x3FFF;
Tmp.Sign = lhs.Sign;
return Tmp;
#endif
}
@@ -284,12 +257,12 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
// Zero is a special case, the exponent is always -inf
if (lhs.Top.Exponent == 0x0 && lhs.Significand == 0x0) {
if (lhs.Exponent == 0x0 && lhs.Significand == 0x0) {
X80SoftFloat Result(1, 0x7FFFUL, 0x8000'0000'0000'0000UL);
return Result;
}
int32_t TrueExp = lhs.Top.Exponent - ExponentBias;
int32_t TrueExp = lhs.Exponent - ExponentBias;
return i32_to_extF80(TrueExp);
#endif
}
@@ -298,11 +271,7 @@ struct FEX_PACKED X80SoftFloat {
FCMP(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs, bool* eq, bool* lt, bool* nan) {
*eq = extF80_eq(state, lhs, rhs);
*lt = extF80_lt(state, lhs, rhs);
// Use IEEE 754 semantics: unordered if neither <, =, nor > is true
// This is more reliable than custom NaN detection
bool gt = !(*eq) && !(*lt) && extF80_le(state, rhs, lhs);
*nan = !(*eq) && !(*lt) && !gt;
*nan = IsNan(lhs) || IsNan(rhs);
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSCALE(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
@@ -421,11 +390,6 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
X80SoftFloat result;
if (HandleInfinityOp(state, lhs, result)) {
return result;
}
BIGFLOAT Src_d = lhs.ToFMax(state);
Src_d = FEXCore::cephes_128bit::tanl(Src_d);
return X80SoftFloat(state, Src_d);
@@ -447,11 +411,6 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
X80SoftFloat result;
if (HandleInfinityOp(state, lhs, result)) {
return result;
}
BIGFLOAT Src_d = lhs.ToFMax(state);
Src_d = FEXCore::cephes_128bit::sinl(Src_d);
return X80SoftFloat(state, Src_d);
@@ -473,11 +432,6 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
X80SoftFloat result;
if (HandleInfinityOp(state, lhs, result)) {
return result;
}
BIGFLOAT Src_d = lhs.ToFMax(state);
Src_d = FEXCore::cephes_128bit::cosl(Src_d);
return X80SoftFloat(state, Src_d);
@@ -505,12 +459,12 @@ struct FEX_PACKED X80SoftFloat {
float ToF32(softfloat_state* state) const {
const float32_t Result = extF80_to_f32(state, *this);
return std::bit_cast<float>(Result);
return FEXCore::BitCast<float>(Result);
}
double ToF64(softfloat_state* state) const {
const float64_t Result = extF80_to_f64(state, *this);
return std::bit_cast<double>(Result);
return FEXCore::BitCast<double>(Result);
}
FEXCore::VectorRegType ToVector() const {
@@ -522,7 +476,7 @@ struct FEX_PACKED X80SoftFloat {
BIGFLOAT ToFMax(softfloat_state* state) const {
#if BIGFLOATSIZE == 16
const float128_t Result = extF80_to_f128(state, *this);
return std::bit_cast<BIGFLOAT>(Result);
return FEXCore::BitCast<BIGFLOAT>(Result);
#else
BIGFLOAT result {};
memcpy(&result, this, sizeof(result));
@@ -576,22 +530,23 @@ struct FEX_PACKED X80SoftFloat {
X80SoftFloat(extFloat80_t rhs) {
Significand = rhs.signif;
Top.Raw = rhs.signExp;
Exponent = rhs.signExp & 0x7FFF;
Sign = rhs.signExp >> 15;
}
X80SoftFloat(softfloat_state* state, const float rhs) {
*this = f32_to_extF80(state, std::bit_cast<float32_t>(rhs));
*this = f32_to_extF80(state, FEXCore::BitCast<float32_t>(rhs));
}
X80SoftFloat(softfloat_state* state, const double rhs) {
*this = f64_to_extF80(state, std::bit_cast<float64_t>(rhs));
*this = f64_to_extF80(state, FEXCore::BitCast<float64_t>(rhs));
}
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
#if BIGFLOATSIZE == 16
*this = f128_to_extF80(state, std::bit_cast<float128_t>(rhs));
*this = f128_to_extF80(state, FEXCore::BitCast<float128_t>(rhs));
#else
*this = std::bit_cast<long double>(rhs);
*this = FEXCore::BitCast<long double>(rhs);
#endif
}
@@ -609,7 +564,8 @@ struct FEX_PACKED X80SoftFloat {
void operator=(extFloat80_t rhs) {
Significand = rhs.signif;
Top.Raw = rhs.signExp;
Exponent = rhs.signExp & 0x7FFF;
Sign = rhs.signExp >> 15;
}
operator FEXCore::VectorRegType() const {
@@ -619,36 +575,27 @@ struct FEX_PACKED X80SoftFloat {
operator extFloat80_t() const {
extFloat80_t Result {};
Result.signif = Significand;
Result.signExp = Top.Raw;
Result.signExp = Exponent | (Sign << 15);
return Result;
}
static bool IsNan(const X80SoftFloat& lhs) {
return (lhs.Top.Exponent == 0x7FFF) && (lhs.Significand & IntegerBit) && (lhs.Significand & Bottom62Significand);
return (lhs.Exponent == 0x7FFF) && (lhs.Significand & IntegerBit) && (lhs.Significand & Bottom62Significand);
}
static bool SignBit(const X80SoftFloat& lhs) {
return lhs.Top.Sign;
return lhs.Sign;
}
private:
static constexpr uint64_t IntegerBit = (1ULL << 63);
static constexpr uint64_t Bottom62Significand = ((1ULL << 62) - 1);
static constexpr uint32_t ExponentBias = 16383;
// Helper function to check for infinity and set invalid operation flag.
// Returns true if infinity is dealt with, false otherwise.
FEXCORE_PRESERVE_ALL_ATTR static bool HandleInfinityOp(softfloat_state* state, const X80SoftFloat& arg, X80SoftFloat& result) {
if (arg.Top.Exponent == 0x7FFF && arg.Significand == 0x8000000000000000ULL) {
state->exceptionFlags |= softfloat_flag_invalid;
// Return QNaN.
result.Top.Sign = 0;
result.Top.Exponent = 0x7FFF;
result.Significand = 0xC000000000000000ULL;
return true;
}
return false;
}
};
#ifndef _WIN32
static_assert(sizeof(X80SoftFloat) == 10, "tword must be 10bytes in size");
#else
// Padding on this extends to 16-bytes rather than 10-bytes on WIN32.
static_assert(sizeof(X80SoftFloat) == 16, "tword must be 16bytes in size");
#endif
+59 -12
View File
@@ -2,27 +2,74 @@
#pragma once
#include <FEXCore/fextl/string.h>
#include <concepts>
#include <cstdint>
#include <string_view>
#include <optional>
namespace FEXCore::StrConv {
template<std::integral T>
bool Conv(std::string_view Value, T* Result) {
if constexpr (std::is_signed_v<T>) {
*Result = static_cast<T>(std::strtoll(Value.data(), nullptr, 0));
} else {
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
}
[[maybe_unused]]
static bool Conv(std::string_view Value, bool* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
template<typename T, typename = std::enable_if_t<std::is_enum_v<T>, T>>
bool Conv(std::string_view Value, T* Result) {
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
[[maybe_unused]]
static bool Conv(std::string_view Value, uint8_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, fextl::string* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, int8_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint16_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, int16_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint32_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, int32_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint64_t* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, int64_t* Result) {
*Result = std::strtoll(Value.data(), nullptr, 0);
return true;
}
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
[[maybe_unused]]
static bool Conv(std::string_view Value, T* Result) {
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, fextl::string* Result) {
*Result = Value;
return true;
}
+3 -5
View File
@@ -1,11 +1,9 @@
// SPDX-License-Identifier: MIT
#pragma once
#ifdef ARCHITECTURE_x86_64
#ifdef _M_X86_64
#include <xmmintrin.h>
#include <immintrin.h>
#else
#include <cstdint>
#endif
namespace FEXCore {
@@ -13,7 +11,7 @@ struct VectorScalarF64Pair {
double val[2];
};
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
// Can't use uint8x16_t directly from arm_neon.h here.
// Overrides softfloat-3e's defines which causes problems.
using VectorRegType = __attribute__((neon_vector_type(16))) uint8_t;
@@ -25,7 +23,7 @@ static inline VectorRegPairType MakeVectorRegPair(VectorRegType low, VectorRegTy
return VectorRegPairType {low, high};
}
#elif defined(ARCHITECTURE_x86_64)
#elif defined(_M_X86_64)
using VectorRegType = __m128i;
using VectorRegPairType = __m256i;
+22 -23
View File
@@ -30,14 +30,14 @@ class Context;
}
namespace FEXCore::Config {
namespace detail {
namespace DefaultValues {
#define P(x) x
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
#include <FEXCore/Config/ConfigValues.inl>
} // namespace detail
} // namespace DefaultValues
enum Paths {
PATH_DATA_DIR_LOCAL = 0,
@@ -134,7 +134,7 @@ public:
void Load();
template<typename T>
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<StringArrayType, T>)
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<DefaultValues::Type::StringArrayType, T>)
std::optional<T> GetConv(ConfigOption Option) {
const auto it = OptionMap.find(Option);
if (it == OptionMap.end()) {
@@ -142,7 +142,7 @@ public:
}
const auto& Value = it->second;
LOGMAN_THROW_A_FMT(!std::holds_alternative<StringArrayType>(Value), "Tried to get config of invalid type!");
LOGMAN_THROW_A_FMT(!std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
if (std::holds_alternative<T>(Value)) [[likely]] {
return std::get<T>(Value);
@@ -165,7 +165,7 @@ public:
private:
void MergeConfigMap(const LayerOptions& Options);
void MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value);
void MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value);
};
void MetaLayer::Load() {
@@ -181,7 +181,7 @@ void MetaLayer::Load() {
}
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value) {
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value) {
// Environment variables need a bit of additional work
// We want to merge the arrays rather than overwrite entirely
auto MetaEnvironment = OptionMap.find(Option);
@@ -193,7 +193,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Stri
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
const auto AddToMap = [&LookupMap](const StringArrayType& Value) {
const auto AddToMap = [&LookupMap](const DefaultValues::Type::StringArrayType& Value) {
for (const auto& EnvVar : Value) {
const auto ItEq = EnvVar.find_first_of('=');
if (ItEq == fextl::string::npos) {
@@ -209,7 +209,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Stri
}
};
AddToMap(std::get<StringArrayType>(MetaEnvironment->second));
AddToMap(std::get<DefaultValues::Type::StringArrayType>(MetaEnvironment->second));
AddToMap(Value);
// Now with the two layers merged in the map
@@ -225,8 +225,8 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
// Insert this layer's options, overlaying previous options that exist here
for (auto& it : Options) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
LOGMAN_THROW_A_FMT(std::holds_alternative<StringArrayType>(it.second), "Tried to get config of invalid type!");
MergeEnvironmentVariables(it.first, std::get<StringArrayType>(it.second));
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(it.second), "Tried to get config of invalid type!");
MergeEnvironmentVariables(it.first, std::get<DefaultValues::Type::StringArrayType>(it.second));
} else {
OptionMap.insert_or_assign(it.first, it.second);
}
@@ -307,10 +307,12 @@ constexpr char ContainerManager[] = "/run/host/container-manager";
fextl::string FindContainer() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::string Manager {};
fextl::vector<char> Manager {};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
return FEXCore::StringUtils::Trim(Manager);
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
return ManagerStr;
}
}
return {};
@@ -319,10 +321,12 @@ fextl::string FindContainer() {
fextl::string FindContainerPrefix() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::string Manager {};
fextl::vector<char> Manager {};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
if (FEXCore::StringUtils::Trim(Manager) == "pressure-vessel") {
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
// We are running inside of pressure vessel
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
return "/run/host/";
@@ -419,7 +423,7 @@ bool Exists(ConfigOption Option) {
return Meta->OptionExists(Option);
}
std::optional<StringArrayType*> All(ConfigOption Option) {
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
return Meta->All(Option);
}
@@ -432,12 +436,6 @@ std::optional<T> GetConv(ConfigOption Option) {
return Meta->GetConv<T>(Option);
}
template std::optional<bool> GetConv(ConfigOption Option);
template std::optional<uint8_t> GetConv(ConfigOption Option);
template std::optional<int32_t> GetConv(ConfigOption Option);
template std::optional<uint32_t> GetConv(ConfigOption Option);
template std::optional<uint64_t> GetConv(ConfigOption Option);
void Set(ConfigOption Option, std::string_view Data) {
Meta->Set(Option, Data);
}
@@ -493,12 +491,13 @@ template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t De
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
template<typename T>
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List) {
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List) {
auto Value = FEXCore::Config::All(Option);
List->clear();
if (Value) {
*List = **Value;
}
}
template void Value<StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
template void Value<DefaultValues::Type::StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option,
DefaultValues::Type::StringArrayType* List);
} // namespace FEXCore::Config
+81 -129
View File
@@ -4,6 +4,7 @@
"Multiblock": {
"Type": "bool",
"Default": "true",
"ShortArg": "m",
"Desc": [
"Controls multiblock code compilation",
"Can cause long JIT compilation times and stutter"
@@ -12,22 +13,20 @@
"MaxInst": {
"Type": "int32",
"Default": "5000",
"ShortArg": "n",
"Desc": [
"Maximum number of instruction to store in a block"
]
},
"EnableCodeCachingWIP": {
"Type": "bool",
"Default": "false",
"CacheObjectCodeCompilation": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
"TextDefault": "none",
"Choices": [ "none", "read", "readwrite" ],
"ArgumentHandler": "CacheObjectCodeHandler",
"Desc": [
"Enable the code caching subsystem"
]
},
"EnableCodeCacheValidation": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enable expensive validation when loading code caches"
"Cache JIT object code to drive.",
"Allows JIT code to be shared between applications"
]
},
"HostFeatures": {
@@ -69,13 +68,7 @@
"ENABLESVEBITPERM": "enablesvebitperm",
"DISABLESVEBITPERM": "disablesvebitperm",
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
"DISABLEPRESERVEALLABI": "disablepreserveallabi",
"ENABLEWFXT": "enablewfxt",
"DISABLEWFXT": "disablewfxt",
"ENABLE3DNOW": "enable3dnow",
"DISABLE3DNOW": "disable3dnow",
"ENABLESSE4A": "enablesse4a",
"DISABLESSE4A": "disablesse4a"
"DISABLEPRESERVEALLABI": "disablepreserveallabi"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
@@ -96,10 +89,7 @@
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it"
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
]
},
"SmallTSCScale": {
@@ -108,40 +98,28 @@
"Desc": [
"Scales the cycle counter on systems that have low frequencies."
]
},
"HideHybrid": {
"Type": "bool",
"Default": "true",
"Desc": [
"Hides hybrid CPU core arrangement."
]
},
"CPUFeatureRegisters": {
"Type": "str",
"Default": "",
"Desc": [
"Allows overriding cpu feature flags for manual testing"
]
}
},
"Emulation": {
"RootFS": {
"Type": "str",
"Default": "",
"ShortArg": "R",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
"\teg: ~/RootFS/Debian_x86_64",
"Or this can be a name of a rootfs",
"If the named rootfs exists in the FEX data folder then it will use that one",
"\teg: $XDG_DATA_HOME/fex-emu/RootFS/<RootFS name>/",
"If XDG_DATA_HOME is unset, ~/.local/share will be used in its place.",
"\teg: $HOME/.local/share/fex-emu/RootFS/<RootFS name>/"
"\teg: $HOME/.fex-emu/RootFS/<RootFS name>/",
"Or if you have XDG_DATA_HOME the config will search in that directory",
"\teg: $XDG_DATA_HOME/.fex-emu/RootFS/<RootFS name>/"
]
},
"ThunkHostLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
]
@@ -149,6 +127,7 @@
"ThunkGuestLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
@@ -156,20 +135,22 @@
"ThunkConfig": {
"Type": "str",
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks.",
"This can be a filesystem path",
"\teg: ~/MyThunkConfig.json",
"Or this can be a named of a Thunk config file",
"If the named config file exists in the FEX data folder folder the it will use that one",
"\teg: $XDG_DATA_HOME/fex-emu/ThunkConfigs/<ThunkConfig name>",
"If XDG_DATA_HOME is unset, ~/.local/share will be used in its place.",
"\teg: $HOME/.local/share/fex-emu/ThunkConfigs/<ThunkConfig name>"
"\teg: $HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>",
"Or if you have XDG_DATA_HOME the config will search in that directory",
"\teg: $XDG_DATA_HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>"
]
},
"Env": {
"Type": "strarray",
"Default": "",
"ShortArg": "E",
"Desc": [
"Adds an environment variable to the emulated environment."
]
@@ -177,6 +158,7 @@
"HostEnv": {
"Type": "strarray",
"Default": "",
"ShortArg": "H",
"Desc": [
"Adds an environment variable to the host environment.",
"This can be useful for setting environment variables that thunks can pick up.",
@@ -189,50 +171,13 @@
"Desc": [
"Allows the user to pass additional arguments to the application"
]
},
"DisableL2Cache": {
"Type": "bool",
"Default": "false",
"Desc": [
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
"Can potentially introduce more stutters."
]
},
"DynamicL1Cache": {
"Type": "bool",
"Default": "false",
"Desc": [
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
"Can potentially introduce more stutters."
]
},
"DynamicL1CacheIncreaseCountHeuristic": {
"Type": "uint64",
"Default": "250",
"Desc": [
"Threshold of lookups per second that the L1 dynamic cache should increase its size.",
"Lower numbers means more aggressive scaling upward to the maximum size.",
"Higher numbers means more conservative scaling, using less memory.",
"Can potentially introduce stutters, more likely the higher the number.",
"Don't have this number smaller than the decrease count!"
]
},
"DynamicL1CacheDecreaseCountHeuristic": {
"Type": "uint64",
"Default": "50",
"Desc": [
"Threshold of lookups per second that the L1 dynamic cache should decrease its size.",
"The higher the number, the more aggressively it reduces the L1 cache size.",
"Lower numbers means more conservative memory savings.",
"Can potentially introduce more stutters, more likely the higher the number.",
"Don't have this number larger than the increase count!"
]
}
},
"Debug": {
"SingleStep": {
"Type": "bool",
"Default": "false",
"ShortArg": "S",
"Desc": [
"Single stepping configuration."
]
@@ -240,6 +185,7 @@
"GdbServer": {
"Type": "bool",
"Default": "false",
"ShortArg": "G",
"Desc": [
"Enables the GDB server."
]
@@ -273,6 +219,7 @@
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"ShortArg": "g",
"Desc": [
"When the test harness ends, print the GPR state."
]
@@ -280,6 +227,7 @@
"O0": {
"Type": "bool",
"Default": "false",
"ShortArg": "O0",
"Desc": [
"Disables optimizations passes for debugging."
]
@@ -341,21 +289,13 @@
"STATS": "stats"
},
"Desc": [
"Allows controlling of the vixl disassembler for generated ARM code.",
"Allows controlling of the vixl disassembler.",
"\toff: No disassembly will be output",
"\tdispatcher: Will enable disassembly of the JIT dispatcher loop",
"\tblocks: Will enable disassembly of the translated instruction code blocks",
"\tstats: Will print stats when disassembling the code"
]
},
"X86Disassemble": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enables x86/x86-64 guest disassembly output for compiled blocks.",
"Requires FEX to be built with -DENABLE_ZYDIS=TRUE"
]
},
"ForceSVEWidth": {
"Type": "uint32",
"Default": "0",
@@ -377,6 +317,7 @@
"SilentLog": {
"Type": "bool",
"Default": "true",
"ShortArg": "s",
"Desc": [
"Disables logging"
]
@@ -384,9 +325,10 @@
"OutputLog": {
"Type": "str",
"Default": "server",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
"[stderr, server, <Filename>]"
"[stdout, stderr, server, <Filename>]"
]
},
"TelemetryDirectory": {
@@ -394,7 +336,7 @@
"Default": "",
"Desc": [
"Redirects the telemetry folder that FEX usually writes to.",
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/fex-emu/Telemetry/}"
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
]
},
"ProfileStats": {
@@ -404,13 +346,6 @@
"Enables FEX's low-overhead sampling profile statistics.",
"Requires a supported version of Mangohud to see the results"
]
},
"EnableGpuvisProfiling": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enables profiling when FEX was built with the gpuvis profiler backend."
]
}
},
"Hacks": {
@@ -465,11 +400,12 @@
"This is required to ensure a split-lock doesn't tear inside the process"
]
},
"KernelUnalignedAtomicBackpatching": {
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
"Desc": [
"When the kernel unaligned atomic handler is enabled, use backpatching to reduce kernel context switches."
"Automatically enables TSO when shared memory is used.",
"Should work without issues in most cases."
]
},
"VolatileMetadata": {
@@ -477,7 +413,7 @@
"Default": "true",
"Desc": [
"Use volatile metadata in PE files to inform TSO instructions when available.",
"When metadata is unavailable falls back to the currently enabled TSO options."
"When metadata is unavailable falls back to the currently enabled TSO options."
]
},
"X87ReducedPrecision": {
@@ -487,6 +423,23 @@
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
]
},
"ABILocalFlags": {
"Type": "bool",
"Default": "false",
"Desc": [
"When enabled enables an optimization around flags.",
"Assumes flags are not used across cals.",
"Hand-written assembly can violate this assumption."
]
},
"ParanoidTSO": {
"Type": "bool",
"Default": "false",
"Desc": [
"Makes TSO operations even more strict.",
"Forces vector loadstores to also become atomic."
]
},
"StallProcess": {
"Type": "bool",
"Default": "false",
@@ -517,16 +470,32 @@
"Desc": [
"Contrains the startup sleep to only apply to processes that match this name."
]
},
"MonoHacks": {
"Type": "bool",
"Default": "true",
"Desc": [
"Permits a hook-based SMC approach and smaller JIT blocks when mono is detected."
]
}
},
"Misc": {
"AOTIRCapture": {
"Type": "bool",
"Default": "false",
"Desc": [
"Captures IR and generates an AOT IR cache.",
"Captures both the loaded executable and libraries it loads."
]
},
"AOTIRGenerate": {
"Type": "bool",
"Default": "false",
"Desc": [
"Scans file for executable code and generates an AOT IR cache.",
"Does not run the executable."
]
},
"AOTIRLoad": {
"Type": "bool",
"Default": "false",
"Desc": [
"Loads an AOT IR cache for the loaded executable."
]
},
"ServerSocketPath": {
"Type": "str",
"Default": "",
@@ -540,32 +509,15 @@
"Desc": [
"Disables inline syscalls in order to support seccomp handling"
]
},
"ExtendedVolatileMetadata": {
"Type": "str",
"Default": "",
"Desc": [
"Configuration provided volatile metadata. Only implemented for WoW64/arm64ec.",
"Limited in its use but can be handy.",
"Extends on top of what Microsoft has for volatile metadata, but also supported for WoW64.",
"Colon delimited modules, then semi-colon delimited instructions, then comma delimited ranges",
"Default disables TSO in the module, unless instructions overlap the range",
"<module>;<offset begin>-<offset-end>,...;<instruction offset to force TSO>,...:<another>",
"examples:",
" * Disable TSO for a full module: Just provide the module name:",
" `hl2_linux`",
" * Disable TSO for a part of the module:",
" `hl2_linux;<offset begin>-<offset-end>`",
" * Disable TSO for a part of the module, but enable TSO for some instructions within the module",
" `hl2_linux;<offset begin>-<offset-end>;<instruction offset>,<instruction offset>`",
" * Disable TSO for multiple modules",
" `hl2_linux:libsdl2.so`"
]
}
}
},
"UnnamedOptions": {
"Misc": {
"IS_INTERPRETER": {
"Type": "bool",
"Default": "false"
},
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false"
+8 -3
View File
@@ -1,7 +1,6 @@
// SPDX-License-Identifier: MIT
#include "Interface/Context/Context.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Core/CoreState.h>
@@ -9,12 +8,18 @@
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Core/Thunks.h>
#include "FEXCore/Debug/InternalThreadState.h"
#include <string.h>
#include <utility>
namespace FEXCore::Context {
void InitializeStaticTables(OperatingMode Mode) {
X86Tables::InitializeInfoTables(Mode);
IR::InstallOpcodeHandlers(Mode);
}
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext(const FEXCore::HostFeatures& Features) {
return fextl::make_unique<FEXCore::Context::ContextImpl>(Features);
}
+109 -127
View File
@@ -2,54 +2,65 @@
#pragma once
#include "Common/JitSymbols.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include <Interface/IR/IntrusiveIRList.h>
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/AOTIR.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <stdint.h>
#include <atomic>
#include <cstddef>
#include <cstdint>
#include <mutex>
#include <optional>
#include <shared_mutex>
namespace FEXCore {
class SignalDelegator;
class CodeLoader;
class ThunkHandler;
struct LookupCacheWriteLockToken;
namespace Core {
struct DebugData;
struct InternalThreadState;
} // namespace Core
namespace CodeSerialize {
class CodeObjectSerializeService;
}
namespace CPU {
class Arm64JITCore;
class Dispatcher;
} // namespace CPU
namespace HLE {
class SourcecodeResolver;
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
} // namespace HLE
} // namespace FEXCore
namespace FEXCore::IR {
struct IRListCopy;
class IRListView;
namespace Validation {
class IRValidation;
}
} // namespace FEXCore::IR
namespace FEXCore::Context {
struct FEX_PACKED ExitFunctionLinkData {
uint64_t HostCode;
uint64_t HostBranch;
uint64_t GuestRIP;
int64_t CallerOffset;
};
struct CustomIRResult {
@@ -61,66 +72,10 @@ struct CustomIRResult {
, Data(Data) {}
};
using BlockDelinkerFunc = void (*)(FEXCore::Context::ExitFunctionLinkData* Record);
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
class CodeCache : public AbstractCodeCache {
public:
CodeCache(ContextImpl&);
~CodeCache();
ContextImpl& CTX;
fextl::unique_ptr<ContextImpl> ValidationCTX;
fextl::unique_ptr<Core::InternalThreadState> ValidationThread;
FEXCore::Core::CPUState::gdt_segment ValidationGDT[32] {};
bool IsGeneratingCache = false;
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
uint64_t ComputeCodeMapId(std::string_view Filename, int FD) override;
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
bool LoadData(Core::InternalThreadState*, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
/**
* Performs expensive extra validation on the loaded code cache data.
*
* This kicks off an in-process recompile of all cached blocks and compares
* them with the cached data. Differences will be reported as fatal errors,
* which can uncover bugs like for example:
* - mismatches of the JIT configuration used during cache generation
* - hidden position dependencies due to missing FEX relocations
* - incorrect instruction padding
*/
void Validate(const ExecutableFileSectionInfo&, fextl::set<uint64_t> GuestBlocks, const fextl::set<uint64_t>& HostBlocks,
std::span<std::byte> CachedCode);
void InitiateCacheGeneration() override {
IsGeneratingCache = true;
}
/**
* Applies a set of FEX relocations to the given code section.
*
* FEX relocations describe runtime-dependencies of FEX-generated code.
* When loading a code cache, they are used to move cached code to the
* dynamically chosen base address of the guest binary.
*
* Conversely, relocations are applied in reverse when writing code caches
* to ensure consistency across generation runs.
*
* Note that FEX relocations are unrelated to ELF/PE relocations.
*
* @param GuestDelta Guest address offset to apply to RIP-relative data
* @param ForStorage True for serializing data (producing deterministic output); false for de-serializing it (resolving dynamic symbols)
*
* @return Returns true on success
*/
[[nodiscard]]
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations, bool ForStorage);
};
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
public:
// Context base class implementation.
bool InitCore() override;
@@ -134,7 +89,6 @@ public:
bool IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) override;
bool IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) override;
uint64_t GetGuestBlockEntry(FEXCore::Core::InternalThreadState* Thread) override;
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs, uint64_t PSTATE) override;
@@ -190,28 +144,35 @@ public:
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
CodeCache& GetCodeCache() override {
return CodeCache;
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
}
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
}
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
}
void SetCodeMapWriter(fextl::unique_ptr<CodeMapWriter> Writer) override {
CodeMapWriter = std::move(Writer);
void FinalizeAOTIRCache() override {
IRCaptureCache.FinalizeAOTIRCache();
}
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void FlushAndCloseCodeMap() override {
if (CodeMapWriter) {
CodeMapWriter.reset();
}
}
void OnCodeBufferAllocated(const std::shared_ptr<CPU::CodeBuffer>&) override;
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
void InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) override;
void InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
return CodeInvalidationMutex;
}
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
@@ -226,13 +187,14 @@ public:
void RemoveForceTSOInformation(uint64_t Address, uint64_t Size) override;
void MarkMonoDetected() override {
MonoDetected = true;
}
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
public:
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
friend class FEXCore::IR::Validation::IRValidation;
struct {
uint64_t VirtualMemSize {1ULL << 36};
uint64_t TSCScale = 0;
@@ -245,8 +207,13 @@ public:
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
@@ -254,12 +221,13 @@ public:
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
} Config;
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
@@ -273,21 +241,37 @@ public:
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
FEXCore::ThunkHandler* ThunkHandler {};
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CodeCache CodeCache;
fextl::unique_ptr<CodeMapWriter> CodeMapWriter;
SignalDelegator* SignalDelegation {};
X86GeneratedCode X86CodeGen;
ContextImpl(const FEXCore::HostFeatures& Features);
~ContextImpl();
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
// This is used as a replacement for the SMC writes in the mono callsite backpatcher that avoids atomic operations
// (safe as the invalidation mutex is locked) and manually invalidates the modified range. Allowing SMC to be detected
// even if faulting is disabled.
static void MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint8_t Size, uint64_t Address, uint64_t Value);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, ExitFunctionLinkData* Record) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
void RemoveCustomIREntrypoint(FEXCore::Core::InternalThreadState* Thread, uintptr_t Entrypoint);
return Fn(Frame, Record);
}
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
// Must be called from owning thread
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
// NOTE: Other threads sharing the same CodeBuffer may reference
// invalidated data ranges through their L1/L2 caches. This is
// not currently a problem since FEX does not repurpose the
// invalidated CodeBuffer memory range currently.
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
struct GenerateIRResult {
std::optional<IR::IRListView> IRView;
@@ -295,28 +279,30 @@ public:
uint64_t TotalInstructionsLength;
uint64_t StartAddr;
uint64_t Length;
bool NeedsAddGuestCodeRanges;
};
[[nodiscard]]
GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
struct CompileCodeResult {
CPU::CPUBackend::CompiledCode CompiledCode;
void* CompiledCode;
fextl::unique_ptr<FEXCore::Core::DebugData> DebugData;
uint64_t StartAddr;
uint64_t Length;
bool NeedsAddGuestCodeRanges;
};
[[nodiscard]]
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
IR::OpSize GetGPROpSize() const {
return Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
}
FEXCore::JITSymbols Symbols;
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator {"FEXMem_OpDispatcher"};
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator {"FEXMem_Frontend"};
FEXCore::Utils::PooledAllocatorVirtualWithGuard CPUBackendAllocator {"FEXMem_CPUBackend"};
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
FEXCore::Utils::PooledAllocatorVirtual CPUBackendAllocator;
// If Atomic-based TSO emulation is enabled or not.
bool IsAtomicTSOEnabled() const {
@@ -346,10 +332,6 @@ public:
return ExitOnHLT;
}
bool AreMonoHacksActive() const {
return Config.MonoHacks && MonoDetected;
}
protected:
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
@@ -357,10 +339,17 @@ protected:
AtomicTSOEmulationEnabled = false;
VectorAtomicTSOEmulationEnabled = false;
MemcpyAtomicTSOEmulationEnabled = false;
} else if (Config.ParanoidTSO) {
AtomicTSOEmulationEnabled = true;
VectorAtomicTSOEmulationEnabled = true;
MemcpyAtomicTSOEmulationEnabled = true;
} else {
AtomicTSOEmulationEnabled = Config.TSOEnabled;
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
MemcpyAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.MemcpySetTSOEnabled;
// Atomic TSO emulation only enabled if the config option is enabled.
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
}
}
@@ -374,6 +363,10 @@ private:
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
bool IsMemoryShared = false;
bool SupportsHardwareTSO = false;
bool AtomicTSOEmulationEnabled = true;
bool VectorAtomicTSOEmulationEnabled = false;
@@ -384,19 +377,8 @@ private:
std::shared_mutex CustomIRMutex;
std::atomic<bool> HasCustomIRHandlers {};
struct CustomIRHandlerEntry final {
CustomIREntrypointHandler Handler;
void* Creator;
void* Data;
};
fextl::unordered_map<uint64_t, CustomIRHandlerEntry> CustomIRHandlers;
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void*, void*>> CustomIRHandlers;
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
fextl::set<uint64_t> ForceTSOInstructions;
bool MonoDetected = false;
std::atomic<uint64_t> MonoBackpatcherBlock;
std::mutex CodeBufferListLock;
fextl::vector<std::weak_ptr<CPU::CodeBuffer>> CodeBufferList;
};
} // namespace FEXCore::Context
+47 -35
View File
@@ -7,11 +7,12 @@
namespace FEXCore::IR {
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
Ref Tmp = A.Base;
if (A.Offset) {
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Offset) : IREmit->Constant(A.Offset);
Ref Offset = IREmit->_Constant(A.Offset);
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, Offset) : Offset;
}
if (A.Index) {
@@ -21,10 +22,10 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
if (Tmp) {
Tmp = IREmit->_AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
} else {
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->Constant(Log2));
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->_Constant(Log2));
}
} else {
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Index) : A.Index;
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Index) : A.Index;
}
}
@@ -40,28 +41,32 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
} else if (A.Offset) {
uint64_t X = A.Offset;
X &= (1ull << Bits) - 1;
Tmp = IREmit->Constant(X);
Tmp = IREmit->_Constant(X);
}
}
if (A.Segment && AddSegmentBase) {
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Segment) : A.Segment;
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Segment) : A.Segment;
}
return Tmp ?: IREmit->Constant(0);
return Tmp ?: IREmit->_Constant(0);
}
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
bool Vector, IR::OpSize AccessSize) {
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
IR::OpSize AccessSize) {
auto SoftwareAddressCalculation = [IREmit, &A, GPRSize]() -> AddressMode {
return {
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
.Index = IREmit->Invalid(),
};
};
const auto Is32Bit = GPRSize == OpSize::i32Bit;
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
if (!GPRSizeMatchesAddrSize || OffsetIndexToLargeFor32Bit) {
// If address size doesn't match GPR size then no optimizations can occur.
return {
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
.Index = IREmit->Invalid(),
};
return SoftwareAddressCalculation();
}
// Loadstore rules:
@@ -95,33 +100,43 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
const bool OffsetIsSIMM9 = A.Offset && A.Offset >= -256 && A.Offset <= 255;
const bool OffsetIsUnsignedScaled = A.Offset > 0 && (A.Offset & (AccessSizeAsImm - 1)) == 0 && (A.Offset / AccessSizeAsImm) <= 4095;
if ((AtomicTSO && !Vector && HostSupportsTSOImm9 && OffsetIsSIMM9) || (!AtomicTSO && (OffsetIsSIMM9 || OffsetIsUnsignedScaled))) {
auto InlineImmOffsetLoadstore = [IREmit, &GPRSize](AddressMode A) -> AddressMode {
// Peel off the offset
AddressMode B = A;
B.Offset = 0;
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset),
.IndexType = MemOffsetType::SXTX,
.Index = IREmit->_Constant(A.Offset),
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = 1,
};
}
};
auto ScaledRegisterLoadstore = [IREmit, GPRSize](AddressMode A) -> AddressMode {
if (A.Index && A.Segment) {
A.Base = IREmit->_Add(GPRSize, A.Base, A.Segment);
} else if (A.Segment) {
A.Index = A.Segment;
A.IndexScale = 1;
}
return A;
};
if (AtomicTSO) {
// TODO: LRCPC3 support for vector Imm9.
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
AddressMode B = A;
// ScaledRegisterLoadstore
if (B.Index && B.Segment) {
B.Base = IREmit->Add(GPRSize, B.Base, B.Segment);
} else if (B.Segment) {
B.Index = B.Segment;
B.IndexScale = 1;
if (!Vector) {
if (HostSupportsTSOImm9 && OffsetIsSIMM9) {
return InlineImmOffsetLoadstore(A);
}
} else {
// TODO: LRCPC3 support for vector Imm9.
}
} else {
if (OffsetIsSIMM9 || OffsetIsUnsignedScaled) {
return InlineImmOffsetLoadstore(A);
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
return ScaledRegisterLoadstore(A);
}
return B;
}
if (Vector || !AtomicTSO) {
@@ -135,8 +150,8 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset),
.IndexType = MemOffsetType::SXTX,
.Index = IREmit->_Constant(A.Offset),
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = 1,
};
}
@@ -144,10 +159,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
}
// Fallback on software address calculation
return {
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
.Index = IREmit->Invalid(),
};
return SoftwareAddressCalculation();
}
+6 -7
View File
@@ -11,18 +11,17 @@ struct AddressMode {
Ref Segment {nullptr};
Ref Base {nullptr};
Ref Index {nullptr};
int64_t Offset = 0;
MemOffsetType IndexType = MemOffsetType::SXTX;
MemOffsetType IndexType = MEM_OFFSET_SXTX;
uint8_t IndexScale = 1;
int64_t Offset = 0;
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
IR::OpSize AddrSize;
bool NonTSO;
};
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
bool Vector, IR::OpSize AccessSize);
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
IR::OpSize AccessSize);
} // namespace FEXCore::IR
}; // namespace FEXCore::IR
@@ -1,10 +1,10 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "FEXCore/Core/X86Enums.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Context/Context.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
@@ -41,7 +41,7 @@ namespace FEXCore::CPU {
// r19-r29 and SP.
namespace x64 {
#ifndef ARCHITECTURE_arm64ec
#ifndef _M_ARM_64EC
// All but x19 and x29 are caller saved
// Note that rax/rdx are rearranged here so we can coalesce cmpxchg.
constexpr std::array<ARMEmitter::Register, 18> SRA = {
@@ -73,13 +73,13 @@ namespace x64 {
ARMEmitter::Reg::r8, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
};
constexpr std::array<ARMEmitter::Register, 7> RA = {
constexpr std::array<ARMEmitter::Register, 8> RA = {
// All these callee saved
ARMEmitter::Reg::r20, ARMEmitter::Reg::r21, ARMEmitter::Reg::r22, ARMEmitter::Reg::r23,
ARMEmitter::Reg::r24, ARMEmitter::Reg::r30, ARMEmitter::Reg::r18,
ARMEmitter::Reg::r24, ARMEmitter::Reg::r25, ARMEmitter::Reg::r30, ARMEmitter::Reg::r18,
};
constexpr unsigned RAPairs = 4;
constexpr unsigned RAPairs = 6;
// Dynamic GPRs
constexpr std::array<ARMEmitter::Register, 2> PreserveAll_Dynamic = {
@@ -143,18 +143,18 @@ namespace x64 {
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r8,
};
constexpr std::array<ARMEmitter::Register, 6> RA = {
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15, ARMEmitter::Reg::r16, ARMEmitter::Reg::r30,
constexpr std::array<ARMEmitter::Register, 7> RA = {
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15,
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
};
constexpr std::array<ARMEmitter::Register, 5> PreserveAll_Dynamic = {ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r16,
ARMEmitter::Reg::r17, ARMEmitter::Reg::r30};
constexpr std::array<ARMEmitter::Register, 5> PreserveAll_Dynamic = {
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
};
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = {ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14,
ARMEmitter::Reg::r15, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
ARMEmitter::Reg::r30};
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = RA;
constexpr unsigned RAPairs = 4;
constexpr unsigned RAPairs = 6;
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
ARMEmitter::VReg::v0, ARMEmitter::VReg::v1, ARMEmitter::VReg::v2, ARMEmitter::VReg::v3,
@@ -245,12 +245,14 @@ namespace x32 {
REG_AF,
};
constexpr std::array<ARMEmitter::Register, 14> RA = {
constexpr std::array<ARMEmitter::Register, 15> RA = {
// All these callee saved
ARMEmitter::Reg::r20,
ARMEmitter::Reg::r21,
ARMEmitter::Reg::r22,
ARMEmitter::Reg::r23,
ARMEmitter::Reg::r24,
ARMEmitter::Reg::r25,
// Registers only available on 32-bit
// All these are caller saved (except for r19).
@@ -263,7 +265,6 @@ namespace x32 {
ARMEmitter::Reg::r29,
ARMEmitter::Reg::r30,
ARMEmitter::Reg::r24,
ARMEmitter::Reg::r19,
};
@@ -272,7 +273,7 @@ namespace x32 {
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
};
constexpr unsigned RAPairs = 10;
constexpr unsigned RAPairs = 12;
// All are caller saved
constexpr std::array<ARMEmitter::VRegister, 8> SRAFPR = {
@@ -369,8 +370,6 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
// Hardcode a 256-bit vector width if we are running in the simulator.
// Allow the user to override this.
Simulator.SetVectorLengthInBits(ForceSVEWidth() ? ForceSVEWidth() : 256);
// FEX doesn't support GCS.
Simulator.DisableGCSCheck();
#endif
#ifdef VIXL_DISASSEMBLER
// Only setup the disassembler if enabled.
@@ -417,34 +416,18 @@ FEXCore::X86State::X86Reg Arm64Emitter::GetX86RegRelationToARMReg(ARMEmitter::Re
return FEXCore::X86State::X86Reg::REG_INVALID;
}
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad, int MaxBytes) {
bool NOPPad = false;
if (Pad == PadType::DOPAD) {
NOPPad = true;
} else if (Pad == PadType::NOPAD) {
NOPPad = false;
} else if (Pad == PadType::AUTOPAD) {
// Force NOP padding to ensure relocated constants always have enough encoding space available
NOPPad = EnableCodeCaching;
}
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
const auto UpperBound = Is64Bit ? 4 : 2;
int Segments = MaxBytes ? (MaxBytes / 2) : UpperBound;
LOGMAN_THROW_A_FMT(MaxBytes >= 0 && MaxBytes <= (UpperBound * 2) && (MaxBytes & 1) == 0,
"MaxBytes must be bounded in the range of [0, {}] and 16-bit aligned", UpperBound);
// If MaxBytes specified then make sure to sanity check incoming data.
LOGMAN_THROW_A_FMT(MaxBytes == 0 || (Constant >> (MaxBytes * 8)) == 0, "MaxBytes provided but data can't fit within provided range.");
int Segments = Is64Bit ? 4 : 2;
if (Is64Bit && ((~Constant) >> 16) == 0) {
movn(s, Reg, (~Constant) & 0xFFFF);
if (NOPPad) {
nop();
nop();
nop();
}
movn(s, Reg, (~Constant) & 0xFFFF);
return;
}
@@ -452,17 +435,17 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
// If the upper 32-bits is all zero, we can now switch to a 32-bit move.
s = ARMEmitter::Size::i32Bit;
Is64Bit = false;
Segments = std::min(Segments, 2);
Segments = 2;
}
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
movn(s, Reg.W(), (~Constant) & 0xFFFF);
if (NOPPad) {
nop();
nop();
nop();
}
movn(s, Reg.W(), (~Constant) & 0xFFFF);
return;
}
@@ -483,24 +466,24 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
// `movz` is better than `orr` since hardware will rename or merge if possible when `movz` is used.
const auto IsImm = ARMEmitter::Emitter::IsImmLogical(Constant, RegSizeInBits(s));
if (IsImm) {
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
if (NOPPad) {
nop();
nop();
nop();
}
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
return;
}
}
// If we can't handle negatives with the orr, try with movn+movk
if (Is64Bit && ((~Constant) >> 32) == 0) {
movn(s, Reg, (~Constant) & 0xFFFF);
movk(s, Reg, (Constant >> 16) & 0xFFFF, 16);
if (NOPPad) {
nop();
nop();
}
movn(s, Reg, (~Constant) & 0xFFFF);
movk(s, Reg, (Constant >> 16) & 0xFFFF, 16);
return;
}
@@ -512,7 +495,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
uint64_t AlignedPC = PC & ~0xFFFULL;
// Offset from aligned PC
auto AlignedOffset = std::bit_cast<int64_t>(Constant - AlignedPC);
int64_t AlignedOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(AlignedPC);
int NumMoves = 0;
@@ -528,7 +511,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
} else {
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
// 21-bit signed integer here
auto SmallOffset = std::bit_cast<int64_t>(Constant - PC);
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
if (ARMEmitter::Emitter::IsInt21(SmallOffset)) {
adr(Reg, SmallOffset);
} else {
@@ -677,7 +660,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
}
}
void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask, bool NZCV) {
void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
#ifndef VIXL_SIMULATOR
if (EmitterCTX->HostFeatures.SupportsAFP) {
// Disable AFP features when spilling registers.
@@ -698,23 +681,19 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
}
#endif
if (NZCV) {
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
// is always static and almost certainly clobbered by the subsequent code.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
}
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
// is always static and almost certainly clobbered by the subsequent code.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
GPRSpillMask &= ~PFAFSpillMask;
str(REG_CALLRET_SP, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.callret_sp));
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i + 1];
@@ -728,9 +707,9 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
}
// Now handle PF/AF
if (NZCV && PFAFSpillMask) {
if (PFAFSpillMask) {
auto PFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
[[maybe_unused]] auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
LOGMAN_THROW_A_FMT(AFOffset == PFOffset + 4, "PF/AF are together");
@@ -778,7 +757,7 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
}
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask, std::optional<ARMEmitter::Register> OptionalReg,
std::optional<ARMEmitter::Register> OptionalReg2, bool NZCV) {
std::optional<ARMEmitter::Register> OptionalReg2) {
auto FindTempReg = [this](uint32_t* GPRFillMask) -> std::optional<ARMEmitter::Register> {
for (auto Reg : StaticRegisters) {
if (((1U << Reg.Idx()) & *GPRFillMask)) {
@@ -804,23 +783,19 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
auto TmpReg = *OptionalReg;
auto TmpReg2 = *OptionalReg2;
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
// Load STATE in from the CPU area as x28 is not callee saved in the ARM64EC ABI.
ldr(TmpReg.X(), ARMEmitter::Reg::r18, TEB_CPU_AREA_OFFSET);
ldr(STATE, TmpReg, CPU_AREA_EMULATOR_DATA_OFFSET);
#endif
ldr(REG_CALLRET_SP, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.callret_sp));
if (NZCV) {
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
// is always static and was almost certainly clobbered.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
}
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
// is always static and was almost certainly clobbered.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
FillSpecialRegs(TmpReg, TmpReg2, true, FPRs);
@@ -881,7 +856,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
}
// Now handle PF/AF
if (NZCV && PFAFFillMask) {
if (PFAFFillMask) {
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
ldp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
@@ -1,38 +1,36 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Config/Config.h>
#include "FEXCore/Utils/EnumUtils.h"
#include "Interface/Core/ObjectCache/Relocations.h"
#ifdef VIXL_DISASSEMBLER
#include <aarch64/disasm-aarch64.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/vector.h>
#endif
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#include <aarch64/simulator-constants-aarch64.h>
#endif
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/vector.h>
#include <CodeEmitter/Emitter.h>
#include <CodeEmitter/Registers.h>
#include <cstddef>
#include <cstdint>
#include <optional>
#include <span>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::X86State {
enum X86Reg : uint32_t;
}
namespace FEXCore::CPU {
// Contains the address to the currently available CPU state
constexpr auto STATE = ARMEmitter::XReg::x28;
#ifndef ARCHITECTURE_arm64ec
#ifndef _M_ARM_64EC
// GPR temporaries. Only x3 can be used across spill boundaries
// so if these ever need to change, be very careful about that.
constexpr auto TMP1 = ARMEmitter::XReg::x0;
@@ -45,8 +43,6 @@ constexpr bool TMP_ABIARGS = true;
constexpr auto REG_PF = ARMEmitter::Reg::r26;
constexpr auto REG_AF = ARMEmitter::Reg::r27;
constexpr auto REG_CALLRET_SP = ARMEmitter::XReg::x25;
// Vector temporaries
constexpr auto VTMP1 = ARMEmitter::VReg::v0;
constexpr auto VTMP2 = ARMEmitter::VReg::v1;
@@ -65,8 +61,6 @@ constexpr bool TMP_ABIARGS = false;
constexpr auto REG_PF = ARMEmitter::Reg::r9;
constexpr auto REG_AF = ARMEmitter::Reg::r24;
constexpr auto REG_CALLRET_SP = ARMEmitter::XReg::x17;
// Vector temporaries
constexpr auto VTMP1 = ARMEmitter::VReg::v16;
constexpr auto VTMP2 = ARMEmitter::VReg::v17;
@@ -90,8 +84,7 @@ constexpr uint64_t EC_CODE_BITMAP_MAX_ADDRESS = 1ULL << 47;
#endif
// Will force one single instruction block to be generated first if set when entering the JIT filling SRA.
// FillStaticRegs must preserve this
constexpr auto ENTRY_FILL_SRA_SINGLE_INST_REG = TMP2;
constexpr auto ENTRY_FILL_SRA_SINGLE_INST_REG = TMP1;
// Predicate to use in the X87 SVE optimization
constexpr ARMEmitter::PRegister PRED_X87_SVEOPT = ARMEmitter::PReg::p2;
@@ -106,20 +99,9 @@ constexpr ARMEmitter::PRegister PRED_TMP_32B = ARMEmitter::PReg::p7;
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
class Arm64Emitter : public ARMEmitter::Emitter {
public:
protected:
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
enum class PadType {
// Explicitly does not need padding, even if code-caching is enabled.
NOPAD,
// Explicitly needs padding, even if code-caching is disabled.
DOPAD,
// Choose to pad or not depending on if code-caching is enabled.
AUTOPAD,
};
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad = PadType::NOPAD, int MaxBytes = 0);
protected:
FEXCore::Context::ContextImpl* EmitterCTX;
std::span<const ARMEmitter::Register> StaticRegisters {};
@@ -129,16 +111,18 @@ protected:
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
uint32_t PairRegisters = 0;
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
// Correlate an ARM register back to an x86 register index.
// Returning REG_INVALID if there was no mapping.
FEXCore::X86State::X86Reg GetX86RegRelationToARMReg(ARMEmitter::Register Reg);
void SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U, bool NZCV = true);
void SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U,
std::optional<ARMEmitter::Register> OptionalReg = std::nullopt,
std::optional<ARMEmitter::Register> OptionalReg2 = std::nullopt, bool NZCV = true);
std::optional<ARMEmitter::Register> OptionalReg2 = std::nullopt);
// Register 0-18 + 29 + 30 are caller saved
static constexpr uint32_t CALLER_GPR_MASK = 0b0110'0000'0000'0111'1111'1111'1111'1111U;
@@ -281,8 +265,6 @@ protected:
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
#endif
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
};
} // namespace FEXCore::CPU
+24 -35
View File
@@ -1,18 +1,14 @@
// SPDX-License-Identifier: MIT
#include "FEXCore/Config/Config.h"
#include "FEXCore/IR/IR.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/Utils/PrctlUtils.h>
#include <cstdint>
#include "LookupCache.h"
#ifndef _WIN32
#include <linux/prctl.h>
#include <sys/prctl.h>
#endif
@@ -278,37 +274,37 @@ namespace CPU {
: ThreadState(ThreadState)
, CodeBuffers(CodeBuffers) {
auto& Ptrs = ThreadState->CurrentFrame->Pointers;
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
// Initialize named vector constants.
for (size_t i = 0; i < FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX; ++i) {
Ptrs.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
}
// Copy named vector constants.
memcpy(Ptrs.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
// Initialize Indexed named vector constants.
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] =
reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] =
reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] =
reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] =
reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] =
reinterpret_cast<uint64_t>(DPPS_MASK.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] =
reinterpret_cast<uint64_t>(DPPD_MASK.data());
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] =
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] =
reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
#ifndef FEX_DISABLE_TELEMETRY
// Fill in telemetry values
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
auto& Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
Ptrs.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(&Telem);
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(&Telem);
}
#endif
}
@@ -321,7 +317,7 @@ namespace CPU {
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
RegisterForSignalHandler(std::move(PrevCodeBuffer));
RegisterForSignalHandler(PrevCodeBuffer);
return CurrentCodeBuffer.get();
}
@@ -330,13 +326,14 @@ namespace CPU {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Keep a reference to the old code buffer to delay deallocation
SignalHandlerCodeBuffers.push_back(std::move(CodeBuffer));
SignalHandlerCodeBuffers.push_back(CodeBuffer);
} else {
SignalHandlerCodeBuffers.clear();
}
}
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
fextl::shared_ptr<CodeBuffer> OldCodeBuffer;
auto NewCodeBuffer = CodeBuffers.GetLatest();
if (CurrentCodeBuffer != NewCodeBuffer) {
RegisterForSignalHandler(CurrentCodeBuffer);
@@ -350,7 +347,7 @@ namespace CPU {
}
CodeBuffer::CodeBuffer(size_t Size)
: AllocatedSize(Size) {
: Size(Size) {
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
@@ -361,13 +358,11 @@ namespace CPU {
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
}
FEXCore::Allocator::VirtualName("FEXMemJIT", reinterpret_cast<void*>(Ptr), Size);
LookupCache = fextl::make_unique<GuestToHostMap>();
}
CodeBuffer::~CodeBuffer() {
FEXCore::Allocator::VirtualFree(Ptr, AllocatedSize);
FEXCore::Allocator::VirtualFree(Ptr, Size);
}
auto CodeBufferManager::AllocateNew(size_t Size) -> fextl::shared_ptr<CodeBuffer> {
@@ -401,20 +396,14 @@ namespace CPU {
Latest = Buffer;
LatestOffset = 0;
OnCodeBufferAllocated(Buffer);
OnCodeBufferAllocated(*Buffer);
return Buffer;
}
fextl::shared_ptr<CodeBuffer> CodeBufferManager::GetLatest() {
if (!Latest) {
if (FEXCore::Config::Get_ENABLECODECACHINGWIP()) {
// Start with a larger code buffer to avoid resizes that would discard
// code loaded from caches
AllocateNew(MAX_CODE_SIZE);
} else {
AllocateNew(INITIAL_CODE_SIZE);
}
AllocateNew(INITIAL_CODE_SIZE);
}
return Latest;
}
@@ -425,7 +414,7 @@ namespace CPU {
return GetLatest();
}
auto NewCodeBufferSize = GetLatest()->AllocatedSize;
auto NewCodeBufferSize = GetLatest()->Size;
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
return AllocateNew(NewCodeBufferSize);
}
@@ -435,7 +424,7 @@ namespace CPU {
auto CheckCodeBuffer = [](CodeBuffer& Buffer, uintptr_t Address) {
// The last page of the code buffer is protected, so we need to exclude it from the valid range
// when checking if the address is in the code buffer.
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.AllocatedSize - 1, FEXCore::Utils::FEX_PAGE_SIZE);
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
return (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr);
};
+23 -14
View File
@@ -13,14 +13,9 @@ $end_info$
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <FEXCore/fextl/map.h>
#include <cstdint>
namespace FEXCore::CPU {
union Relocation;
}
namespace FEXCore {
namespace IR {
@@ -43,7 +38,7 @@ struct GuestToHostMap;
namespace CPU {
struct CodeBuffer {
uint8_t* Ptr;
size_t AllocatedSize; // including guard page; see UsableSize()
size_t Size;
fextl::unique_ptr<GuestToHostMap> LookupCache;
@@ -54,11 +49,6 @@ namespace CPU {
CodeBuffer& operator=(CodeBuffer&&) = delete;
~CodeBuffer();
/// Returns the number of bytes available for storing code
size_t UsableSize() const {
return AllocatedSize - FEXCore::Utils::FEX_PAGE_SIZE;
}
};
/**
@@ -86,7 +76,7 @@ namespace CPU {
// Protects writes to the latest CodeBuffer and changes to LatestOffset
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
virtual void OnCodeBufferAllocated(CodeBuffer&) {};
private:
fextl::shared_ptr<CodeBuffer> Latest;
@@ -104,7 +94,15 @@ namespace CPU {
struct CompiledCode {
// Where this code block begins.
uint8_t* BlockBegin;
fextl::map<uint64_t, uint8_t*> EntryPoints;
/**
* The function entrypoint to this codeblock.
*
* This may or may not equal `BlockBegin` above. Depending on the CPU backend, it may stick data
* prior to the BlockEntry.
*
* Is actually a function pointer of type `void (FEXCore::Core::ThreadState *Thread)`
*/
uint8_t* BlockEntry;
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
size_t Size;
};
@@ -166,7 +164,18 @@ namespace CPU {
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) = 0;
/**
* @brief Relocates a block of code from the JIT code object cache
*
* @param Entry - RIP of the entry
* @param SerializationData - Serialization data referring to the object cache for `Entry`
*
* @return An executable function pointer relocated from the cache object
*/
[[nodiscard]]
virtual void* RelocateJITObjectCode(uint64_t /* Entry */, const CodeSerialize::CodeObjectFileSection* /* SerializationData */) {
return nullptr;
}
virtual void ClearCache() {}
+140 -169
View File
@@ -14,7 +14,6 @@ $end_info$
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Utils/FileLoading.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/Syscalls.h>
@@ -24,7 +23,7 @@ $end_info$
namespace FEXCore {
namespace ProductNames {
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
static const char ARM_UNKNOWN[] = "Unknown ARM CPU";
static const char ARM_A57[] = "Cortex-A57";
static const char ARM_A72[] = "Cortex-A72";
@@ -44,15 +43,12 @@ namespace ProductNames {
static const char ARM_A715[] = "Cortex-A715";
static const char ARM_A720[] = "Cortex-A720";
static const char ARM_A725[] = "Cortex-A725";
static const char ARM_C1Pro[] = "C1-Pro";
static const char ARM_C1Premium[] = "C1-Premium";
static const char ARM_X1[] = "Cortex-X1";
static const char ARM_X1C[] = "Cortex-X1C";
static const char ARM_X2[] = "Cortex-X2";
static const char ARM_X3[] = "Cortex-X3";
static const char ARM_X4[] = "Cortex-X4";
static const char ARM_X925[] = "Cortex-X925";
static const char ARM_C1Ultra[] = "C1-Ultra";
static const char ARM_N1[] = "Neoverse N1";
static const char ARM_N2[] = "Neoverse N2";
static const char ARM_N3[] = "Neoverse N3";
@@ -63,7 +59,6 @@ namespace ProductNames {
static const char ARM_A65[] = "Cortex-A65";
static const char ARM_A510[] = "Cortex-A510";
static const char ARM_A520[] = "Cortex-A520";
static const char ARM_C1Nano[] = "C1-Nano";
static const char ARM_Kryo200[] = "Kryo 2xx";
static const char ARM_Kryo300[] = "Kryo 3xx";
@@ -75,7 +70,6 @@ namespace ProductNames {
static const char ARM_Denver[] = "Nvidia Denver";
static const char ARM_Carmel[] = "Nvidia Carmel";
static const char ARM_Olympus[] = "Nvidia Olympus";
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
@@ -89,13 +83,9 @@ namespace ProductNames {
static const char ARM_Blizzard_M2Pro[] = "Apple Blizzard (M2 Pro)";
static const char ARM_Avalanche_M2Max[] = "Apple Avalanche (M2 Max)";
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
static const char ARM_AppleSilicon[] = "Apple Silicon";
static const char ARM_ORYON_1[] = "Oryon-1";
static const char ARM_Ampere_1[] = "AmpereOne";
static const char ARM_Ampere_1A[] = "AmpereOneA";
static const char ARM_Ampere_1B[] = "AmpereOneB";
static const char ARM_Ampere_1C[] = "AmpereOneC";
#else
#endif
} // namespace ProductNames
@@ -140,7 +130,7 @@ constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
});
#endif
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
uint32_t GetCycleCounterFrequency() {
uint64_t Result {};
__asm("mrs %[Res], CNTFRQ_EL0" : [Res] "=r"(Result));
@@ -154,7 +144,6 @@ uint32_t GetCPUID_TPIDRRO() {
}
void CPUIDEmu::SetupHostHybridFlag() {
FEX_CONFIG_OPT(HideHybrid, HIDEHYBRID);
PerCPUData.resize(Cores);
uint64_t MIDR {};
@@ -171,11 +160,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
MIDR = NewMIDR;
}
if (HideHybrid()) {
// Hide the hybrid flag.
Hybrid = false;
}
struct CPUMIDR {
uint8_t Implementer;
uint16_t Part;
@@ -186,7 +170,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
// CPU priority order
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
// Relative list so things they will commonly end up in big.little configurations sort of relate
static constexpr std::array<CPUMIDR, 67> CPUMIDRs = {{
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
// Typically big CPU cores
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
@@ -196,49 +180,39 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x029, 1, ProductNames::ARM_Firestorm_M1Max}, // Apple Firestorm (M1 Max)
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
{0x61, 0, 1, ProductNames::ARM_AppleSilicon}, // QEmu Apple Silicon
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
{0x41, 0xd8b, 1, ProductNames::ARM_C1Pro}, // C1-Pro
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0xc0, 0xac3, 1, ProductNames::ARM_Ampere_1}, // AmpereOne
{0xc0, 0xac4, 1, ProductNames::ARM_Ampere_1A}, // AmpereOneA
{0xc0, 0xac5, 1, ProductNames::ARM_Ampere_1B}, // AmpereOneB
{0xc0, 0xac7, 1, ProductNames::ARM_Ampere_1C}, // AmpereOneC
{0x4e, 0x010, 1, ProductNames::ARM_Olympus}, // Olympus
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
// Denver rated above A57 to match TX2 weirdness
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
@@ -253,7 +227,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
{0x41, 0xd8a, 1, ProductNames::ARM_C1Nano}, // C1-Nano
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
@@ -393,8 +366,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
} else {
// If we aren't hybrid then just claim everything is big
for (size_t i = 0; i < Cores; ++i) {
const auto MIDRIndex = HideHybrid() ? 0 : i;
uint32_t MIDR = PerCPUData[MIDRIndex].MIDR;
uint32_t MIDR = PerCPUData[i].MIDR;
auto MIDROption = FindDefinedMIDR(MIDR);
PerCPUData[i].IsBig = true;
@@ -452,10 +424,10 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
Res.eax = FAMILY_IDENTIFIER;
Res.ebx = 0 | // Brand index
(8 << 8) | // Cache line size in bytes
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(GetCPUID() << 24); // Local APIC ID
Res.ebx = 0 | // Brand index
(8 << 8) | // Cache line size in bytes
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(0 << 24); // Local APIC ID
Res.ecx = (1 << 0) | // SSE3
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
@@ -518,7 +490,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
(1 << 25) | // SSE
(1 << 26) | // SSE2
(0 << 27) | // Self Snoop
(0 << 28) | // (HTT) Max APIC IDs reserved field is valid
(1 << 28) | // Max APIC IDs reserved field is valid
(1 << 29) | // Thermal monitor
(0 << 30) | // Reserved
(0 << 31); // Pending break enable
@@ -654,7 +626,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
// Only enable EnhancedREPMOVS if atomic memcpy tso emulation isn't enabled.
const uint32_t SupportsEnhancedREPMOVS = CTX->IsMemcpyAtomicTSOEnabled() == false;
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT;
// Number of subfunctions
Res.eax = 0x0;
@@ -674,39 +645,39 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(1 << 13) | // Deprecates FPU CS and DS
(0 << 14) | // Intel MPX
(0 << 15) | // Intel Resource Directory Technology Allocation
(0 << 16) | // AVX512-F
(0 << 17) | // AVX512-DQ
(0 << 16) | // Reserved
(0 << 17) | // Reserved
(CTX->HostFeatures.SupportsRAND << 18) | // RDSEED
(1 << 19) | // ADCX and ADOX instructions
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
(0 << 21) | // AVX512-IFMA
(0 << 22) | // PCOMMIT (deprecated?)
(0 << 21) | // Reserved
(0 << 22) | // Reserved
(1 << 23) | // CLFLUSHOPT instruction
(1 << 24) | // CLWB instruction
(0 << 25) | // Intel processor trace
(0 << 26) | // AVX512-PF
(0 << 27) | // AVX512-ER
(0 << 28) | // AVX512-CD
(0 << 26) | // Reserved
(0 << 27) | // Reserved
(0 << 28) | // Reserved
(Features.SHA << 29) | // SHA instructions
(0 << 30) | // AVX512-BW
(0 << 31); // AVX512-VL
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.ecx = (1 << 0) | // PREFETCHWT1
(0 << 1) | // AVX512VBMI
(0 << 2) | // Usermode instruction prevention
(0 << 3) | // Protection keys for user mode pages
(0 << 4) | // OS protection keys
(SupportsWFXT << 5) | // waitpkg
(0 << 6) | // AVX512-VBMI2
(0 << 5) | // waitpkg
(0 << 6) | // AVX512_VBMI2
(0 << 7) | // CET shadow stack
(0 << 8) | // GFNI
(CTX->HostFeatures.SupportsAES256 << 9) | // VAES
(SupportsVPCLMULQDQ << 10) | // VPCLMULQDQ
(0 << 11) | // AVX512-VNNI
(0 << 12) | // AVX512-BITALG
(0 << 11) | // AVX512_VNNI
(0 << 12) | // AVX512_BITALG
(0 << 13) | // Intel Total Memory Encryption
(0 << 14) | // AVX512-VPOPCNTDQ
(0 << 15) | // FZM (TDX)
(0 << 14) | // AVX512_VPOPCNTDQ
(0 << 15) | // Reserved
(0 << 16) | // 5 Level page tables
(0 << 17) | // MPX MAWAU
(0 << 18) | // MPX MAWAU
@@ -714,28 +685,28 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 20) | // MPX MAWAU
(0 << 21) | // MPX MAWAU
(1 << 22) | // RDPID Read Processor ID
(0 << 23) | // AES Key Locker
(1 << 24) | // bus-lock-detect
(0 << 23) | // Reserved
(0 << 24) | // Reserved
(0 << 25) | // CLDEMOTE
(0 << 26) | // MPRR (TDX)
(0 << 26) | // Reserved
(0 << 27) | // MOVDIRI
(0 << 28) | // MOVDIR64B
(0 << 29) | // ENQCMD
(0 << 29) | // Reserved
(0 << 30) | // SGX Launch configuration
(0 << 31); // PKS
(0 << 31); // Reserved
Res.edx = (0 << 0) | // SGX-TEM (TDX)
(0 << 1) | // SGX-KEYS
(0 << 2) | // AVX512-4VNNIW
(0 << 3) | // AVX512-4FMAPS
Res.edx = (0 << 0) | // Reserved
(0 << 1) | // Reserved
(0 << 2) | // AVX512_4VNNIW
(0 << 3) | // AVX512_4FMAPS
(1 << 4) | // Fast Short Rep Mov
(0 << 5) | // UINTR
(0 << 5) | // Reserved
(0 << 6) | // Reserved
(0 << 7) | // Reserved
(0 << 8) | // AVX512-VP2INTERSECT
(0 << 8) | // AVX512_VP2INTERSECT
(0 << 9) | // SRBDS_CTRL (Special Register Buffer Data Sampling Mitigations)
(0 << 10) | // VERW clears CPU buffers
(0 << 11) | // rtm-always-abort
(0 << 11) | // Reserved
(0 << 12) | // Reserved
(0 << 13) | // TSX Force Abort (TSX will force abort if attempted)
(0 << 14) | // SERIALIZE instruction
@@ -747,7 +718,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 20) | // Intel CET
(0 << 21) | // Reserved
(0 << 22) | // AMX-BF16 - Tile computation on bfloat16
(0 << 23) | // AVX512-FP16 - FP16 AVX512 instructions
(0 << 23) | // AVX512_FP16 - FP16 AVX512 instructions
(0 << 24) | // AMX-tile - If AMX is implemented
(0 << 25) | // AMX-int8 - AMX on 8-bit integers
(0 << 26) | // IBRS_IBPB - Speculation control
@@ -783,7 +754,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) const {
// XFeatureSupportedMask[63:32]
Res.edx = 0; // Upper 32-bits of XFeatureSupportedMask
} else if (Leaf == 1) {
Res.eax = (1 << 0) | // XSAVEOPT
Res.eax = (0 << 0) | // XSAVEOPT
(0 << 1) | // XSAVEC (and XRSTOR)
(0 << 2) | // XGETBV - XGETBV with ECX=1 supported
(0 << 3); // XSAVES - XSAVES, XRSTORS, and IA32_XSS supported
@@ -865,10 +836,10 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0001h(uint32_t Leaf) con
constexpr uint32_t MaximumSubLeafNumber = 2;
if (Leaf == 0) {
// EAX[3:0] Is the host architecture that FEX is running under
#ifdef ARCHITECTURE_x86_64
#ifdef _M_X86_64
// EAX[3:0] = 1 = x86_64 host architecture
Res.eax |= 0b0001;
#elif defined(ARCHITECTURE_arm64)
#elif defined(_M_ARM_64)
// EAX[3:0] = 2 = AArch64 host architecture
Res.eax |= 0b0010;
#else
@@ -920,71 +891,71 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
Res.eax = FAMILY_IDENTIFIER;
Res.ecx = (1 << 0) | // LAHF/SAHF
(1 << 1) | // 0 = Single core product, 1 = multi core product
(0 << 2) | // SVM
(1 << 3) | // Extended APIC register space
(0 << 4) | // LOCK MOV CR0 means MOV CR8
(1 << 5) | // ABM instructions
(CTX->HostFeatures.SupportsSSE4a << 6) | // SSE4a
(0 << 7) | // Misaligned SSE mode
(1 << 8) | // PREFETCHW
(0 << 9) | // OS visible workaround support
(0 << 10) | // Instruction based sampling support
(0 << 11) | // XOP
(0 << 12) | // SKINIT
(0 << 13) | // Watchdog timer support
(0 << 14) | // Reserved
(0 << 15) | // Lightweight profiling support
(0 << 16) | // FMA4
(1 << 17) | // Translation cache extension
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(0 << 20) | // Reserved
(0 << 21) | // XOP-TBM
(0 << 22) | // Topology extensions support
(0 << 23) | // Core performance counter extensions
(0 << 24) | // NB performance counter extensions
(0 << 25) | // Reserved
(0 << 26) | // Data breakpoints extensions
(0 << 27) | // Performance TSC
(0 << 28) | // L2 perf counter extensions
(0 << 29) | // MONITORX
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.ecx = (1 << 0) | // LAHF/SAHF
(1 << 1) | // 0 = Single core product, 1 = multi core product
(0 << 2) | // SVM
(1 << 3) | // Extended APIC register space
(0 << 4) | // LOCK MOV CR0 means MOV CR8
(1 << 5) | // ABM instructions
(0 << 6) | // SSE4a
(0 << 7) | // Misaligned SSE mode
(1 << 8) | // PREFETCHW
(0 << 9) | // OS visible workaround support
(0 << 10) | // Instruction based sampling support
(0 << 11) | // XOP
(0 << 12) | // SKINIT
(0 << 13) | // Watchdog timer support
(0 << 14) | // Reserved
(0 << 15) | // Lightweight profiling support
(0 << 16) | // FMA4
(1 << 17) | // Translation cache extension
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(0 << 20) | // Reserved
(0 << 21) | // XOP-TBM
(0 << 22) | // Topology extensions support
(0 << 23) | // Core performance counter extensions
(0 << 24) | // NB performance counter extensions
(0 << 25) | // Reserved
(0 << 26) | // Data breakpoints extensions
(0 << 27) | // Performance TSC
(0 << 28) | // L2 perf counter extensions
(0 << 29) | // MONITORX
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.edx = (1 << 0) | // FPU
(1 << 1) | // Virtual mode extensions
(1 << 2) | // Debugging extensions
(1 << 3) | // Page size extensions
(1 << 4) | // TSC
(1 << 5) | // MSR support
(1 << 6) | // PAE
(1 << 7) | // Machine Check Exception
(1 << 8) | // CMPXCHG8B
(1 << 9) | // APIC
(0 << 10) | // Reserved
(1 << 11) | // SYSCALL/SYSRET
(1 << 12) | // MTRR
(1 << 13) | // Page global extension
(1 << 14) | // Machine Check architecture
(1 << 15) | // CMOV
(1 << 16) | // Page attribute table
(1 << 17) | // Page-size extensions
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(1 << 20) | // NX
(0 << 21) | // Reserved
(1 << 22) | // MMXExt
(1 << 23) | // MMX
(1 << 24) | // FXSAVE/FXRSTOR
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
(0 << 26) | // 1 gigabit pages
(SUPPORTS_RDTSCP << 27) | // RDTSCP
(0 << 28) | // Reserved
(1 << 29) | // Long Mode
(CTX->HostFeatures.Supports3DNow << 30) | // 3DNow! Extensions
(CTX->HostFeatures.Supports3DNow << 31); // 3DNow!
Res.edx = (1 << 0) | // FPU
(1 << 1) | // Virtual mode extensions
(1 << 2) | // Debugging extensions
(1 << 3) | // Page size extensions
(1 << 4) | // TSC
(1 << 5) | // MSR support
(1 << 6) | // PAE
(1 << 7) | // Machine Check Exception
(1 << 8) | // CMPXCHG8B
(1 << 9) | // APIC
(0 << 10) | // Reserved
(1 << 11) | // SYSCALL/SYSRET
(1 << 12) | // MTRR
(1 << 13) | // Page global extension
(1 << 14) | // Machine Check architecture
(1 << 15) | // CMOV
(1 << 16) | // Page attribute table
(1 << 17) | // Page-size extensions
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(1 << 20) | // NX
(0 << 21) | // Reserved
(1 << 22) | // MMXExt
(1 << 23) | // MMX
(1 << 24) | // FXSAVE/FXRSTOR
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
(0 << 26) | // 1 gigabit pages
(SUPPORTS_RDTSCP << 27) | // RDTSCP
(0 << 28) | // Reserved
(1 << 29) | // Long Mode
(1 << 30) | // 3DNow! Extensions
(1 << 31); // 3DNow!
return Res;
}
@@ -1105,9 +1076,9 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) con
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
uint32_t CoreCount = Cores - 1;
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
(std::bit_ceil(Cores) << 12) | // ApicIdSize: Number of bits in ApicID
(CoreCount << 0); // Count count subtract one
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
(CoreCount << 0); // Count count subtract one
return Res;
}
@@ -1238,7 +1209,7 @@ CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
SetupFeatures();
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
if (SupportsCPUIndexInTPIDRRO) {
GetCPUID = GetCPUID_TPIDRRO;
}
+2 -2
View File
@@ -159,7 +159,7 @@ private:
struct CPUData {
const char* ProductName {};
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
uint32_t MIDR {};
#endif
bool IsBig {};
@@ -277,7 +277,7 @@ private:
// 0: Highest function parameter and ID
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
// 1: Processor info
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
// 2: Cache and TLB info
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
// 3: Serial Number(previously), now reserved
-666
View File
@@ -1,666 +0,0 @@
// SPDX-License-Identifier: MIT
#include "Utils/SpinWaitLock.h"
#include <Interface/Context/Context.h>
#include <Interface/Core/ArchHelpers/Arm64Emitter.h>
#include <Interface/Core/Dispatcher/Dispatcher.h>
#include <Interface/Core/JIT/DebugData.h>
#include <Interface/Core/JIT/Relocations.h>
#include <Interface/Core/LookupCache.h>
#include <Interface/Core/OpcodeDispatcher.h>
#include <Interface/IR/PassManager.h>
#include <FEXCore/Core/Thunks.h>
#include <FEXCore/HLE/SourcecodeResolver.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXHeaderUtils/Filesystem.h>
#include <git_version.h>
#include <xxhash.h>
#include <fstream>
namespace FEXCore {
#if __clang_major__ < 16
ExecutableFileInfo::ExecutableFileInfo(fextl::unique_ptr<HLE::SourcecodeMap> Map, uint64_t FileId, fextl::string Filename)
: SourcecodeMap(std::move(Map))
, FileId(FileId)
, Filename(Filename) {}
#endif
ExecutableFileInfo::~ExecutableFileInfo() = default;
fextl::string CodeMap::GetBaseFilename(const ExecutableFileInfo& MainExecutable, bool AddNombSuffix) {
auto FileId = MainExecutable.FileId;
std::string_view base_filename = FHU::Filesystem::GetFilename(std::string_view {MainExecutable.Filename});
if (FileId != 0xffff'ffff'ffff'ffff) {
return fextl::fmt::format("{}-{:016x}{}", base_filename, MainExecutable.FileId, AddNombSuffix ? "-nomb" : "");
}
return "";
}
fextl::map<CodeMapFileId, CodeMap::ParsedContents> CodeMap::ParseCodeMap(std::ifstream& File) {
fextl::map<CodeMapFileId, CodeMap::ParsedContents> Ret;
while (true) {
Entry Entry;
File.read(reinterpret_cast<char*>(&Entry), sizeof(Entry));
if (!File) {
break;
}
if (Entry.FileId == LoadExternalLibrary.FileId && Entry.BlockOffset == LoadExternalLibrary.BlockOffset) {
ExternalLibraryInfo Info;
File.read(reinterpret_cast<char*>(&Info), sizeof(Info));
fextl::string Filename;
std::getline(File, Filename, '\0');
// Align to 4-byte boundary
char Null[4];
File.read(Null, AlignUp(Filename.size() + 1, 4) - Filename.size() - 1);
if (!File) {
break;
}
Ret[Info.ExternalFileId].Filename = std::move(Filename);
} else if (Entry.FileId == SetExecutableFileId {}.Marker.FileId && Entry.BlockOffset == SetExecutableFileId {}.Marker.BlockOffset) {
CodeMapFileId ExecutableFileId;
File.read(reinterpret_cast<char*>(&ExecutableFileId), sizeof(ExecutableFileId));
if (!File) {
break;
}
Ret[ExecutableFileId].IsExecutable = true;
} else {
if (!Ret.contains(Entry.FileId)) {
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
} else {
Ret[Entry.FileId].Blocks.insert(Entry.BlockOffset);
}
}
if (!File) {
break;
}
}
return Ret;
}
CodeMapWriter::CodeMapWriter(CodeMapOpener& Opener, bool OpenEagerly)
: Buffer(4096)
, FileOpener(Opener) {
if (OpenEagerly) {
CodeMapFD = FileOpener.OpenCodeMapFile();
}
}
CodeMapWriter::~CodeMapWriter() {
if (CodeMapFD.value_or(-1) != -1) {
Flush(BufferOffset);
close(*CodeMapFD);
}
}
bool CodeMapWriter::IsWriteEnabled(const ExecutableFileSectionInfo& Section) {
if (CodeMapFD == -1) {
return false;
}
// PV libraries can't yet be read by FEXServer, so skip dumping them
if (Section.FileInfo.Filename.starts_with("/run/pressure-vessel")) {
return false;
}
if (CodeMapFD) {
return true;
}
// Acquire mutex and re-check CodeMapFD to avoid race conditions
auto lk = std::unique_lock {Mutex};
if (!CodeMapFD) {
CodeMapFD = FileOpener.OpenCodeMapFile();
}
return CodeMapFD != -1;
}
void CodeMapWriter::Flush(size_t Offset) {
// Acquire exclusive lock and flush circular buffer
std::unique_lock Lock {Mutex};
Flush(Offset, Lock);
}
void CodeMapWriter::Flush(size_t Offset, std::unique_lock<std::shared_mutex>&) {
write(*CodeMapFD, Buffer.data(), Offset);
BufferOffset = 0;
}
void CodeMapWriter::AppendBlock(const FEXCore::ExecutableFileSectionInfo& SectionInfo, uint64_t BlockEntry) {
if (!IsWriteEnabled(SectionInfo)) {
return;
}
BlockEntry -= SectionInfo.FileStartVA;
if (BlockEntry > std::numeric_limits<uint32_t>::max()) {
ERROR_AND_DIE_FMT("Cannot write code map");
}
// Register new library if not already known
bool NewLibraryLoad = false;
{
// Check prior registration with shared lock
std::shared_lock Lock {Mutex};
NewLibraryLoad = !KnownFileIds.contains(SectionInfo.FileInfo.FileId);
}
if (NewLibraryLoad) {
// Register to map with exclusive lock
std::unique_lock Lock {Mutex};
NewLibraryLoad &= KnownFileIds.insert(SectionInfo.FileInfo.FileId).second;
}
if (NewLibraryLoad) {
// Add entry to code map
AppendLibraryLoad(SectionInfo.FileInfo);
}
// Register the actual code block
CodeMap::Entry DataEntry {SectionInfo.FileInfo.FileId, static_cast<uint32_t>(BlockEntry)};
AppendData(std::as_bytes(std::span {&DataEntry, 1}));
}
void CodeMapWriter::AppendLibraryLoad(const FEXCore::ExecutableFileInfo& FileInfo) {
// See CodeMap::ExternalLibraryInfo
auto ExternalFileId = FileInfo.FileId;
auto TotalSize = AlignUp(sizeof(CodeMap::LoadExternalLibrary) + sizeof(ExternalFileId) + FileInfo.Filename.size() + 1, 4);
const auto Data = reinterpret_cast<char*>(alloca(TotalSize));
auto WritePtr = std::copy_n(reinterpret_cast<const char*>(&CodeMap::LoadExternalLibrary), sizeof(CodeMap::LoadExternalLibrary), Data);
WritePtr = std::copy_n(reinterpret_cast<const char*>(&ExternalFileId), sizeof(ExternalFileId), WritePtr);
WritePtr = std::copy(FileInfo.Filename.begin(), FileInfo.Filename.end(), WritePtr);
std::fill(WritePtr, Data + TotalSize, 0);
AppendData(std::as_bytes(std::span {Data, TotalSize}));
}
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo) {
CodeMap::SetExecutableFileId Data {.ExecutableFileId = FileInfo.FileId};
AppendData(std::span {reinterpret_cast<const std::byte*>(&Data), sizeof(Data)});
}
void CodeMapWriter::AppendData(std::span<const std::byte> Data) {
std::shared_lock Lock {Mutex};
auto Offset = BufferOffset.fetch_add(Data.size_bytes());
if (Offset + Data.size_bytes() > Buffer.size()) {
// Acquire exclusive lock and flush the buffer.
// Under heavy pressure, multiple threads may observe an exhausted buffer simultaneously.
// The thread with the last in-bounds Offset is responsible for flushing the buffer.
Lock.unlock();
bool IsResponsibleForFlush = false;
{
std::unique_lock ExclusiveLock {Mutex};
IsResponsibleForFlush = (Offset <= Buffer.size());
if (IsResponsibleForFlush) {
Flush(Offset, ExclusiveLock);
}
}
if (!IsResponsibleForFlush) {
// Wait for the buffer to be flushed on the responsible thread
Utils::SpinWaitLock::WaitPred<std::less_equal<>, size_t>(reinterpret_cast<size_t*>(&BufferOffset), Buffer.size());
}
AppendData(Data);
return;
}
memcpy(&Buffer.at(Offset), Data.data(), Data.size_bytes());
}
} // namespace FEXCore
namespace FEXCore::Context {
CodeCache::CodeCache(ContextImpl& CTX_)
: CTX(CTX_) {}
CodeCache::~CodeCache() = default;
uint64_t CodeCache::ComputeCodeMapId(std::string_view Filename, int FD) {
if (Filename.empty()) {
return 0xffff'ffff'ffff'ffff;
}
// For now, we just use the file path as an identifier.
// TODO: Ensure the hash is unique enough to distinguish executables while remaining independent of the installation location
return XXH3_64bits(Filename.data(), Filename.size());
}
struct CodeCacheHeader {
std::array<char, 4> Magic = ExpectedMagic;
uint32_t FormatVersion = 1;
uint8_t FEXVersion[20] = {};
uint32_t NumBlocks;
uint32_t NumCodePages;
uint32_t CodeBufferSize;
uint32_t NumRelocations;
uint32_t padding;
uint64_t SerializedBaseAddress;
// TODO: Consider including information from LookupCache.BlockLinks
static constexpr std::array<char, 4> ExpectedMagic = {'F', 'X', 'C', 'C'};
};
template<typename T>
concept OrderedContainer = requires { typename T::key_compare; };
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
auto CodeBuffer = CTX.GetLatest();
auto& LookupCache = *Thread.LookupCache->Shared;
auto Relocations = Thread.CPUBackend->TakeRelocations(SourceBinary.FileStartVA);
// Write file header
CodeCacheHeader header {};
static_assert(GIT_HASH.size() == sizeof(header.FEXVersion));
std::ranges::copy(GIT_HASH, header.FEXVersion);
header.NumBlocks = LookupCache.BlockList.size();
header.NumCodePages = LookupCache.CodePages.size();
header.CodeBufferSize = CTX.LatestOffset;
header.NumRelocations = Relocations.size();
header.SerializedBaseAddress = SerializedBaseAddress;
::write(fd, &header, sizeof(header));
// Dump guest<->host block mappings
{
// Cache contents must be deterministic, so copy the unordered block list and then sort by key
static_assert(!OrderedContainer<decltype(LookupCache.BlockList)>, "Already deterministic; drop temporary container");
fextl::vector<std::pair<uint64_t, const GuestToHostMap::BlockEntry*>> BlockList;
BlockList.reserve(LookupCache.BlockList.size());
for (auto& [Guest, BlockEntry] : LookupCache.BlockList) {
static_assert(sizeof(Guest) == 8, "Breaking change in code cache data layout");
BlockList.emplace_back(Guest, &BlockEntry);
}
std::ranges::sort(BlockList);
for (auto [Guest, Host] : BlockList) {
static_assert(sizeof(Host->HostCode) == 8, "Breaking change in code cache data layout");
static_assert(sizeof(Host->CodePages[0]) == 8, "Breaking change in code cache data layout");
Guest -= SourceBinary.FileStartVA;
::write(fd, &Guest, sizeof(Guest));
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->Ptr);
::write(fd, &HostCode, sizeof(HostCode));
uint64_t NumCodePages = Host->CodePages.size();
::write(fd, &NumCodePages, sizeof(NumCodePages));
LOGMAN_THROW_A_FMT(std::ranges::is_sorted(Host->CodePages), "Code pages aren't sorted");
for (auto CodePage : Host->CodePages) {
CodePage -= SourceBinary.FileStartVA;
::write(fd, &CodePage, sizeof(CodePage));
}
}
}
// Dump relocations
static_assert(sizeof(Relocations[0]) == 48, "Breaking change in code cache data layout");
::write(fd, Relocations.data(), Relocations.size() * sizeof(Relocations[0]));
// Pad to next page in file so that the CodeBuffer can be mmap'ed into process on load
char Zero[64] {};
auto Off = lseek(fd, 0, SEEK_CUR);
while (Off != AlignUp(Off, Utils::FEX_PAGE_SIZE)) {
auto BytesToWrite = std::min(AlignUp(Off, Utils::FEX_PAGE_SIZE) - Off, sizeof(Zero));
::write(fd, Zero, BytesToWrite);
Off += BytesToWrite;
}
// Dump the host code (relocated for position-independent serialization)
std::span CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->Ptr), reinterpret_cast<std::byte*>(CodeBuffer->Ptr) + CTX.LatestOffset);
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, true)) {
LOGMAN_THROW_A_FMT(false, "Failed to apply code relocations");
return false;
}
::write(fd, CodeBufferData.data(), CodeBufferData.size());
// Dump code pages
static_assert(OrderedContainer<decltype(LookupCache.CodePages)>, "Non-deterministic data source");
for (const auto& [PageIndex, Entrypoints] : LookupCache.CodePages) {
uint64_t PageAddr = (PageIndex << 12) - SourceBinary.FileStartVA;
::write(fd, &PageAddr, sizeof(PageAddr));
uint64_t NumEntrypoints = Entrypoints.size();
::write(fd, &NumEntrypoints, sizeof(NumEntrypoints));
for (uint64_t Entrypoint : Entrypoints) {
Entrypoint -= SourceBinary.FileStartVA;
::write(fd, &Entrypoint, sizeof(Entrypoint));
}
}
return true;
}
bool CodeCache::LoadData(Core::InternalThreadState* Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& BinarySection) {
if (!EnableCodeCaching) {
return true;
}
namespace ranges = std::ranges;
// Read file header
CodeCacheHeader header {};
::memcpy(&header, MappedCacheFile, sizeof(header));
MappedCacheFile += sizeof(header);
LogMan::Msg::IFmt("Cache load: {:5} blocks; base={:#14x}; off={:#9x}-{:#09x}; {:016x} {}", header.NumBlocks, BinarySection.FileStartVA,
BinarySection.BeginVA - BinarySection.FileStartVA, BinarySection.EndVA - BinarySection.FileStartVA,
BinarySection.FileInfo.FileId, BinarySection.FileInfo.Filename);
if (!ranges::equal(header.Magic, header.ExpectedMagic)) {
LogMan::Msg::EFmt("Invalid cache file header");
return false;
}
if (!ranges::equal(header.FEXVersion, GIT_HASH)) {
LogMan::Msg::IFmt("Cache generated from old FEX version {:02x}, current is {:02x}; skipping", fmt::join(header.FEXVersion, ""),
fmt::join(GIT_HASH, ""));
return false;
}
if (header.NumBlocks == 0) {
// Valid caches are never empty
LogMan::Msg::IFmt("Code cache empty, aborting");
return false;
}
// Read guest<->host block mappings
using BlockListEntry = decltype(GuestToHostMap::BlockList)::value_type;
fextl::vector<BlockListEntry> BlockList(header.NumBlocks);
{
for (auto& BlockPtr : BlockList) {
::memcpy(&BlockPtr.first, MappedCacheFile, sizeof(BlockPtr.first));
MappedCacheFile += sizeof(BlockPtr.first);
::memcpy(&BlockPtr.second.HostCode, MappedCacheFile, sizeof(BlockPtr.second.HostCode));
MappedCacheFile += sizeof(BlockPtr.second.HostCode);
uint64_t NumGuestPages;
::memcpy(&NumGuestPages, MappedCacheFile, sizeof(NumGuestPages));
MappedCacheFile += sizeof(NumGuestPages);
BlockPtr.second.CodePages.resize(NumGuestPages);
::memcpy(BlockPtr.second.CodePages.data(), MappedCacheFile, std::span {BlockPtr.second.CodePages}.size_bytes());
MappedCacheFile += std::span {BlockPtr.second.CodePages}.size_bytes();
}
// Consistency check: VMA regions at the top and end should belong to the same file
auto [min_val, max_val] = ranges::minmax_element(BlockList, std::less {}, &decltype(BlockList)::value_type::first);
auto MinBound = CTX.SyscallHandler->LookupExecutableFileSection(Thread, min_val->first + BinarySection.FileStartVA);
auto MaxBound = CTX.SyscallHandler->LookupExecutableFileSection(Thread, max_val->first + BinarySection.FileStartVA);
if (&MinBound->FileInfo != &BinarySection.FileInfo || &MaxBound->FileInfo != &BinarySection.FileInfo) {
ERROR_AND_DIE_FMT("Cached blocks offsets {:#x}-{:#x} out of bounds for guest library {} ({:016x} @ {:#x}) while trying to load "
"section {:#x}-{:#x}!",
min_val->first, max_val->first, BinarySection.FileInfo.Filename, BinarySection.FileInfo.FileId,
BinarySection.FileStartVA, BinarySection.BeginVA, BinarySection.EndVA);
}
// Constrain BlockList to the given ExecutableFileSectionInfo
LOGMAN_THROW_A_FMT(ranges::is_sorted(BlockList, [](auto& a, auto& b) { return a.first < b.first; }), "Expected sorted block list");
auto begin = ranges::lower_bound(BlockList, BinarySection.BeginVA - BinarySection.FileStartVA, std::less {}, &BlockListEntry::first);
auto end =
ranges::upper_bound(begin, BlockList.end(), BinarySection.EndVA - BinarySection.FileStartVA - 1, std::less {}, &BlockListEntry::first);
BlockList.erase(end, BlockList.end());
BlockList.erase(BlockList.begin(), begin);
if (BlockList.empty()) {
// Not an error since there is just no data to load
LogMan::Msg::IFmt("No blocks cached in this range, aborting");
return true;
}
}
// Read relocations
fextl::vector<FEXCore::CPU::Relocation> Relocations(header.NumRelocations, FEXCore::CPU::Relocation::Default());
::memcpy(Relocations.data(), MappedCacheFile, Relocations.size() * sizeof(Relocations[0]));
MappedCacheFile += Relocations.size() * sizeof(Relocations[0]);
// Pad to next page in file, which contains CodeBuffer data
MappedCacheFile = reinterpret_cast<std::byte*>(AlignUp(reinterpret_cast<uintptr_t>(MappedCacheFile), Utils::FEX_PAGE_SIZE));
// Prepare CodeBuffer: Page aligned and big enough to hold all cached data
auto Lock = std::unique_lock {CTX.CodeBufferWriteMutex};
if (Thread) {
if (auto Prev = Thread->CPUBackend->CheckCodeBufferUpdate()) {
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
auto lk = Thread->LookupCache->AcquireWriteLock();
Thread->LookupCache->ChangeGuestToHostMapping(*Prev, *CTX.GetLatest()->LookupCache, lk);
}
}
auto CodeBuffer = CTX.GetLatest();
LOGMAN_THROW_A_FMT(reinterpret_cast<uintptr_t>(CodeBuffer->Ptr) % 0x1000 == 0, "Expected CodeBuffer base to be page-aligned");
const auto Delta = AlignUp(CTX.LatestOffset, 0x1000) - CTX.LatestOffset;
CTX.LatestOffset += Delta;
while (CTX.LatestOffset + header.CodeBufferSize > CodeBuffer->UsableSize()) {
if (Thread) {
CTX.ClearCodeCache(Thread);
CodeBuffer = CTX.GetLatest();
LogMan::Msg::IFmt("Increased code buffer size to {} MiB for cache load", CodeBuffer->AllocatedSize / 1024 / 1024);
} else {
ERROR_AND_DIE_FMT("Cannot extend codebuffer without thread!");
}
}
// Read CodeBuffer data from file. Make sure the destination is page-aligned.
// TODO: Only load the data needed for the selected section
auto CodeBufferRange =
std::as_writable_bytes(std::span {CodeBuffer->Ptr, CodeBuffer->UsableSize()}).subspan(CTX.LatestOffset, header.CodeBufferSize);
::memcpy(CodeBufferRange.data(), MappedCacheFile, header.CodeBufferSize);
MappedCacheFile += header.CodeBufferSize;
CTX.LatestOffset += header.CodeBufferSize;
// Apply FEX relocations
auto Ret = ApplyCodeRelocations(BinarySection.FileStartVA, CodeBufferRange, Relocations, false);
LOGMAN_THROW_A_FMT(Ret == true, "Failed to apply code cache relocations");
{
auto& LookupCache = *CodeBuffer->LookupCache;
auto WriteLock = LookupCache.AcquireWriteLock();
// Register blocks to LookupCache
for (auto& [Guest, Host] : BlockList) {
for (auto& CodePage : Host.CodePages) {
CodePage += BinarySection.FileStartVA;
}
auto HostCode = reinterpret_cast<void*>(Host.HostCode + reinterpret_cast<uintptr_t>(CodeBufferRange.data()));
LookupCache.AddBlockMapping(Guest + BinarySection.FileStartVA, std::move(Host.CodePages), HostCode, WriteLock);
}
// Register loaded code ranges
fextl::vector<uint64_t> Entrypoints;
for (uint32_t i = 0; i < header.NumCodePages; ++i) {
uint64_t CodePage;
memcpy(&CodePage, MappedCacheFile, sizeof(CodePage));
CodePage += BinarySection.FileStartVA;
MappedCacheFile += sizeof(CodePage);
uint64_t NumEntrypoints;
memcpy(&NumEntrypoints, MappedCacheFile, sizeof(NumEntrypoints));
MappedCacheFile += sizeof(NumEntrypoints);
Entrypoints.resize(NumEntrypoints);
memcpy(Entrypoints.data(), MappedCacheFile, NumEntrypoints * sizeof(Entrypoints[0]));
MappedCacheFile += NumEntrypoints * sizeof(Entrypoints[0]);
for (auto& Entrypoint : Entrypoints) {
Entrypoint += BinarySection.FileStartVA;
}
if (LookupCache.AddBlockExecutableRange(Entrypoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE, WriteLock)) {
CTX.SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
}
if (EnableCodeCacheValidation) {
fextl::set<uint64_t> GuestBlocks, HostBlocks;
for (auto& [Guest, Host] : BlockList) {
GuestBlocks.insert(Guest + BinarySection.FileStartVA);
HostBlocks.insert(Host.HostCode);
}
Validate(BinarySection, std::move(GuestBlocks), HostBlocks, CodeBufferRange);
}
return true;
}
void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<uint64_t> GuestBlocks, const fextl::set<uint64_t>& HostBlocks,
std::span<std::byte> CachedCode) {
LOGMAN_THROW_A_FMT(!HostBlocks.empty(), "Tried to validate without any host blocks");
// Skip any cached data before the first host block
CachedCode = CachedCode.subspan(*HostBlocks.begin() - sizeof(CPU::CPUBackend::JITCodeHeader));
if (!ValidationCTX) {
ValidationCTX.reset(static_cast<ContextImpl*>(FEXCore::Context::Context::CreateNewContext(CTX.HostFeatures).release()));
ValidationCTX->SetSignalDelegator(CTX.SignalDelegation);
ValidationCTX->SetSyscallHandler(CTX.SyscallHandler);
ValidationCTX->SetThunkHandler(CTX.ThunkHandler);
if (!ValidationCTX->InitCore()) {
ERROR_AND_DIE_FMT("Failed to create cache load validation context");
}
ValidationThread.reset(ValidationCTX->CreateThread(0, 0, nullptr));
auto Frame = ValidationThread->CurrentFrame;
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_GDT] = &ValidationGDT[0];
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_LDT] = &ValidationGDT[0];
Frame->State.cs_idx = 0;
Frame->State.cs_cached = 0;
if (ValidationCTX->Config.Is64BitMode()) {
ValidationGDT[0].L = 1; // L = Long Mode = 64-bit
ValidationGDT[0].D = 0; // D = Default Operand Size = Reserved
} else {
ValidationGDT[0].L = 0; // L = Long Mode = 32-bit
ValidationGDT[0].D = 1; // D = Default Operand Size = 32-bit
}
}
auto NewCodeBuffer = ValidationCTX->GetLatest();
while (CachedCode.size_bytes() > NewCodeBuffer->UsableSize()) {
ValidationCTX->ClearCodeCache(ValidationThread.get());
NewCodeBuffer = ValidationCTX->GetLatest();
LogMan::Msg::IFmt("Increased cache validation code buffer size to {} MiB", NewCodeBuffer->AllocatedSize / 1024 / 1024);
}
std::span<std::byte> CodeBufferRangeRef =
std::as_writable_bytes(std::span {NewCodeBuffer->Ptr, NewCodeBuffer->Ptr + NewCodeBuffer->UsableSize()}).subspan(0, CachedCode.size_bytes());
while (!GuestBlocks.empty()) {
auto [CompiledBlocks, _, _2, _3, _4] = ValidationCTX->CompileCode(ValidationThread.get(), *GuestBlocks.begin(), 0 /* TODO: Set MaxInst? */);
for (auto& Entry : CompiledBlocks.EntryPoints) {
GuestBlocks.erase(Entry.first);
}
}
// Patch FEX-internal function addresses with values from the main Context to ensure the code blocks are comparable
auto NewRelocations = ValidationThread->CPUBackend->TakeRelocations(Section.FileStartVA);
NewRelocations.erase(std::remove_if(NewRelocations.begin(), NewRelocations.end(), [](const CPU::Relocation& Reloc) {
return Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL && Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
}));
(void)ApplyCodeRelocations(Section.FileStartVA, CodeBufferRangeRef, NewRelocations, false);
if (ValidationCTX->LatestOffset <= CodeBufferRangeRef.size()) {
// Reference compilation produced fewer bytes than our cache, so validation is going to fail.
// Make sure we don't output any garbage bytes though.
CodeBufferRangeRef = CodeBufferRangeRef.subspan(0, ValidationCTX->LatestOffset);
}
auto [Mismatch, _] = std::mismatch(CodeBufferRangeRef.begin(), CodeBufferRangeRef.end(), CachedCode.begin());
if (Mismatch != CodeBufferRangeRef.end()) {
// Align down to instruction size
auto Idx = AlignDown(std::distance(CodeBufferRangeRef.begin(), Mismatch), 4);
auto BlockIt = std::prev(HostBlocks.lower_bound(*HostBlocks.begin() + Idx + 1));
std::optional<uint64_t> GuestBlockAddr;
std::optional<uint64_t> GuestBlockAddrRef;
if (BlockIt != HostBlocks.end()) {
for (int i : {0, 1}) {
std::span Buffer = (i == 0 ? CachedCode : CodeBufferRangeRef);
// Second instruction is always a constant load for relative offset to the (multi)block start
int32_t addr = (*reinterpret_cast<uint32_t*>(&Buffer[*BlockIt - *HostBlocks.begin() + 4]) & 0x3ff'ffe0) << 11;
addr >>= 14;
auto header = reinterpret_cast<CPU::CPUBackend::JITCodeHeader*>(&Buffer[*BlockIt - *HostBlocks.begin() + 4 + addr]);
auto tail = reinterpret_cast<CPU::CPUBackend::JITCodeTail*>(reinterpret_cast<uintptr_t>(header) + header->OffsetToBlockTail);
(i == 0 ? GuestBlockAddr : GuestBlockAddrRef) = tail->RIP - Section.FileStartVA;
LogMan::Msg::EFmt("Recorded rip {}: {:#x} (offset {:#x})", i, tail->RIP, tail->RIP - Section.FileStartVA);
if (i == 1) {
if (tail->RIP >= Section.BeginVA && tail->RIP < Section.EndVA) {
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, _] =
ValidationCTX->GenerateIR(ValidationThread.get(), tail->RIP, false, FEXCore::Config::Get_MAXINST());
fextl::stringstream ss;
FEXCore::IR::Dump(&ss, &*IRView);
LogMan::Msg::EFmt("IR:\n{}", ss.str());
} else {
LogMan::Msg::EFmt("Can't dump IR for out-of-range RIP {:#x}", tail->RIP);
}
}
}
}
fextl::string GuestBlockInfo = "UNKNOWN";
if (GuestBlockAddr) {
GuestBlockInfo = fextl::fmt::format("{:#x}", GuestBlockAddr.value());
}
if (GuestBlockAddr != GuestBlockAddrRef) {
GuestBlockInfo += " (MISMATCH)";
}
ERROR_AND_DIE_FMT("Cache validation failed at offset {:#x}: {:02x} <-> {:02x} (at {} <-> {}, guest block {})", Idx,
fmt::join(CachedCode.subspan(Idx, 4), ""), fmt::join(CodeBufferRangeRef.subspan(Idx, 4), ""),
fmt::ptr(CachedCode.data()), fmt::ptr(CodeBufferRangeRef.data()), GuestBlockInfo);
}
// Reset Context state for next validation
ValidationThread->LookupCache->ClearCache(ValidationThread->LookupCache->AcquireWriteLock());
ValidationCTX->LatestOffset = 0;
LogMan::Msg::IFmt("\tSuccessfully validated cache");
}
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
std::span<const FEXCore::CPU::Relocation> EntryRelocations, bool ForStorage) {
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
Emitter.SetCursorOffset(Reloc.Header.Offset);
switch (Reloc.Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
// Generate a literal so we can place it
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Reloc.NamedSymbolLiteral.Symbol);
Emitter.dc64(Pointer);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// TODO: Pointers are required to fit within 48-bit VA space.
// But forcing 6-byte broke relocations.
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer,
CPU::Arm64Emitter::PadType::DOPAD);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
Emitter.dc64(GuestEntry + Reloc.GuestRIP.GuestRIP);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
uint64_t Pointer = Reloc.GuestRIP.GuestRIP + GuestEntry;
// TODO: Pointers are required to fit within 48-bit VA space.
// But forcing 6-byte broke relocations.
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIP.RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
break;
}
default: ERROR_AND_DIE_FMT("Unknown relocation type {}", ToUnderlying(Reloc.Header.Type));
}
}
return true;
}
} // namespace FEXCore::Context
+213 -247
View File
@@ -9,19 +9,16 @@ $end_info$
*/
#include <cstdint>
#ifdef ZYDIS_DISASSEMBLER
#include <Zydis/Zydis.h>
#endif
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <Interface/GDBJIT/GDBJIT.h>
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
@@ -49,13 +46,13 @@ $end_info$
#include "FEXCore/Utils/SignalScopeGuards.h"
#include <FEXCore/Utils/Threads.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/SHMStats.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <FEXHeaderUtils/TodoDefines.h>
#include <algorithm>
#include <array>
@@ -81,7 +78,10 @@ namespace FEXCore::Context {
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
: HostFeatures {Features}
, CPUID {this}
, CodeCache {*this} {
, IRCaptureCache {this} {
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
CodeObjectCacheService = fextl::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
}
if (!Config.Is64BitMode()) {
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
Config.VirtualMemSize = 1ULL << 32;
@@ -105,6 +105,14 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
UpdateAtomicTSOEmulationConfig();
}
ContextImpl::~ContextImpl() {
{
if (CodeObjectCacheService) {
CodeObjectCacheService->Shutdown();
}
}
}
struct GetFrameBlockInfoResult {
const CPU::CPUBackend::JITCodeHeader* InlineHeader;
const CPU::CPUBackend::JITCodeTail* InlineTail;
@@ -131,11 +139,6 @@ bool ContextImpl::IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* T
return InlineTail && InlineTail->SingleInst;
}
uint64_t ContextImpl::GetGuestBlockEntry(FEXCore::Core::InternalThreadState* Thread) {
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
return InlineTail ? InlineTail->RIP : 0;
}
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
const auto Frame = Thread->CurrentFrame;
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
@@ -346,9 +349,36 @@ bool ContextImpl::InitCore() {
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
// Set up the SignalDelegator config since core is initialized.
SignalDelegation->SetConfig(Dispatcher->MakeSignalDelegatorConfig());
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
.DispatcherBegin = Dispatcher->Start,
.DispatcherEnd = Dispatcher->End,
#if defined(_WIN32) && !defined(ARCHITECTURE_arm64ec)
.AbsoluteLoopTopAddress = Dispatcher->AbsoluteLoopTopAddress,
.AbsoluteLoopTopAddressFillSRA = Dispatcher->AbsoluteLoopTopAddressFillSRA,
.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress,
.SignalHandlerReturnAddressRT = Dispatcher->SignalHandlerReturnAddressRT,
.PauseReturnInstruction = Dispatcher->PauseReturnInstruction,
.ThreadPauseHandlerAddressSpillSRA = Dispatcher->ThreadPauseHandlerAddressSpillSRA,
.ThreadPauseHandlerAddress = Dispatcher->ThreadPauseHandlerAddress,
// Stop handlers.
.ThreadStopHandlerAddressSpillSRA = Dispatcher->ThreadStopHandlerAddressSpillSRA,
.ThreadStopHandlerAddress = Dispatcher->ThreadStopHandlerAddress,
// SRA information.
.SRAGPRCount = Dispatcher->GetSRAGPRCount(),
.SRAFPRCount = Dispatcher->GetSRAFPRCount(),
};
Dispatcher->GetSRAGPRMapping(SignalConfig.SRAGPRMapping);
Dispatcher->GetSRAFPRMapping(SignalConfig.SRAFPRMapping);
// Give this configuration to the SignalDelegator.
SignalDelegation->SetConfig(SignalConfig);
#ifndef _WIN32
#elif !defined(_M_ARM_64EC)
// WOW64 always needs the interrupt fault check to be enabled.
Config.NeedsPendingInterruptFaultCheck = true;
#endif
@@ -366,26 +396,27 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
}
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
// Update the thread pointer for Thunk return to the latest.
Thread->CurrentFrame->Pointers.ThunkCallbackRet = SignalDelegation->GetThunkCallbackRET();
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
if (CodeObjectCacheService) {
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
// Use the thread's object cache ref counter for this
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
}
// If it is the parent thread that died then just leave
// TODO: This doesn't make sense when the parent thread doesn't outlive its children
FEX_TODO("This doesn't make sense when the parent thread doesn't outlive its children");
}
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(this);
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
Thread->CurrentFrame->State.L1Pointer = Thread->LookupCache->GetL1Pointer();
Thread->CurrentFrame->State.L1Mask = Thread->LookupCache->GetScaledL1PointerMask();
Thread->CurrentFrame->Pointers.L2Pointer = Thread->LookupCache->GetPagePointer();
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
Dispatcher->InitThreadPointers(Thread);
@@ -406,7 +437,6 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXC
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
.CTX = this,
};
FEXCore::Allocator::VirtualName("FEXMem_ThreadState", Thread, sizeof(*Thread));
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
Thread->CurrentFrame->State.rip = InitialRIP;
@@ -443,10 +473,6 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
Profiler::PostForkAction(Child);
if (Child) {
if (CodeMapWriter) {
CodeMapWriter->ResetAfterFork();
}
CodeInvalidationMutex.StealAndDropActiveLocks();
if (Config.StrictInProcessSplitLocks) {
StrictSplitLockMutex = 0;
@@ -469,29 +495,28 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
}
#endif
void ContextImpl::OnCodeBufferAllocated(const fextl::shared_ptr<CPU::CodeBuffer>& Buffer) {
void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
if (Config.GlobalJITNaming()) {
Symbols.RegisterJITSpace(Buffer->Ptr, Buffer->AllocatedSize);
}
{
std::scoped_lock lk {CodeBufferListLock};
CodeBufferList.emplace_back(Buffer);
Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
}
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer) {
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
if (CodeObjectCacheService) {
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
// Use the thread's object cache ref counter for this
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
}
if (NewCodeBuffer) {
// Allocate new CodeBuffer + L3 LookupCache and clear L1+L2 caches
Thread->CPUBackend->ClearCache();
} else {
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
auto lk = Thread->LookupCache->AcquireWriteLock();
Thread->LookupCache->ClearCache(lk);
Thread->LookupCache->ClearCache();
}
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
}
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
@@ -520,7 +545,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
if (Handler != CustomIRHandlers.end()) {
TotalInstructions = 1;
TotalInstructionsLength = 1;
Handler->second.Handler(GuestRIP, Thread->OpDispatcher.get());
std::get<0>(Handler->second)(GuestRIP, Thread->OpDispatcher.get());
HasCustomIR = true;
}
}
@@ -532,34 +557,23 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
bool HadDispatchError {false};
bool HadInvalidInst {false};
Thread->FrontendDecoder->DecodeInstructionsAtEntry(Thread, GuestCode, GuestRIP, MaxInst);
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, MaxInst,
[Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Thread, Start, Length);
}
});
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
auto CodeBlocks = &BlockInfo->Blocks;
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
AreMonoHacksActive() && MonoBackpatcherBlock.load(std::memory_order_relaxed) == GuestRIP);
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount);
const auto GPRSize = Thread->OpDispatcher->GetGPROpSize();
#ifdef ZYDIS_DISASSEMBLER
const auto ZydisMachineMode = Config.Is64BitMode ? ZYDIS_MACHINE_MODE_LONG_64 : ZYDIS_MACHINE_MODE_LEGACY_32;
if (FEXCore::Config::Get_X86DISASSEMBLE()) {
const uint64_t DecodedMin = Thread->FrontendDecoder->DecodedMinAddress;
const uint64_t DecodedMax = Thread->FrontendDecoder->DecodedMaxAddress;
LogMan::Msg::IFmt("Guest x86 Begin (RIP={:#x}, {:#x}-{:#x})", GuestRIP, DecodedMin, DecodedMax);
}
#endif
const auto GPRSize = GetGPROpSize();
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
#ifdef ZYDIS_DISASSEMBLER
if (FEXCore::Config::Get_X86DISASSEMBLE() && CodeBlocks->size() > 1) {
LogMan::Msg::IFmt(" Block {} Entry={:#x} NumInsts={}", j, Block.Entry, Block.NumInstructions);
}
#endif
bool BlockInForceTSOValidRange = false;
auto InstForceTSOIt = ForceTSOInstructions.end();
if (ForceTSOValidRanges.Contains({Block.Entry, Block.Entry + Block.Size})) {
@@ -581,7 +595,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
if (InstsInBlock == 0) {
// Special case for an empty instruction block.
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry - GuestRIP));
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
}
for (size_t i = 0; i < InstsInBlock; ++i) {
@@ -591,19 +605,6 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
TableInfo = Block.DecodedInstructions[i].TableInfo;
DecodedInfo = &Block.DecodedInstructions[i];
#ifdef ZYDIS_DISASSEMBLER
if (FEXCore::Config::Get_X86DISASSEMBLE()) {
const uint8_t* InstBytes = reinterpret_cast<const uint8_t*>(InstAddress);
ZydisDisassembledInstruction ZydisInst;
if (ZYAN_SUCCESS(ZydisDisassembleIntel(ZydisMachineMode, InstAddress, InstBytes, DecodedInfo->InstSize, &ZydisInst))) {
LogMan::Msg::IFmt(" {:#x}: {}", InstAddress, ZydisInst.text);
} else {
LogMan::Msg::IFmt(" {:#x}: (decode failed, {} bytes)", InstAddress, DecodedInfo->InstSize);
}
}
#endif
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
// Do a partial register cache flush before every instruction. This
@@ -624,12 +625,11 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
Thread->OpDispatcher->_GuestOpcode(InstAddress - GuestRIP);
}
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL || Block.ForceFullSMCDetection) {
auto ExistingCodePtr = reinterpret_cast<uint8_t*>(Block.Entry + BlockInstructionsLength);
auto InstAddressReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
std::array<uint8_t, 0x10> CodeOriginal;
memcpy(CodeOriginal.data(), ExistingCodePtr, DecodedInfo->InstSize);
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CodeOriginal, InstAddressReg, DecodedInfo->InstSize);
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
auto ExistingCodePtr = reinterpret_cast<uint64_t*>(Block.Entry + BlockInstructionsLength);
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(ExistingCodePtr[0], ExistingCodePtr[1],
(uintptr_t)ExistingCodePtr - GuestRIP, DecodedInfo->InstSize);
auto InvalidateCodeCond = Thread->OpDispatcher->CondJump(CodeChanged);
@@ -639,7 +639,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP));
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
@@ -647,21 +647,15 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
}
if (TableInfo && TableInfo->OpcodeDispatcher.OpDispatch) {
auto Fn = TableInfo->OpcodeDispatcher.OpDispatch;
if (TableInfo && TableInfo->OpcodeDispatcher) {
auto Fn = TableInfo->OpcodeDispatcher;
Thread->OpDispatcher->ResetHandledLock();
Thread->OpDispatcher->ResetDecodeFailure();
IR::ForceTSOMode ForceTSO = IR::ForceTSOMode::NoOverride;
if (BlockInForceTSOValidRange) {
if (InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress) {
ForceTSO = IR::ForceTSOMode::ForceEnabled;
} else {
ForceTSO = IR::ForceTSOMode::ForceDisabled;
}
} else if (DecodedInfo->Flags & X86Tables::DecodeFlags::FLAG_FORCE_TSO) {
ForceTSO = IR::ForceTSOMode::ForceEnabled;
}
IR::ForceTSOMode ForceTSO =
BlockInForceTSOValidRange ?
(InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress ? IR::ForceTSOMode::ForceEnabled :
IR::ForceTSOMode::ForceDisabled) :
IR::ForceTSOMode::NoOverride;
Thread->OpDispatcher->SetForceTSO(ForceTSO);
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
if (Thread->OpDispatcher->HadDecodeFailure()) {
@@ -689,12 +683,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
}
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::INVALID_INST ||
Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::BAD_RELOCATION) {
Thread->OpDispatcher->InvalidOp(DecodedInfo);
} else {
Thread->OpDispatcher->NoExecOp(DecodedInfo);
}
Thread->OpDispatcher->InvalidOp(DecodedInfo);
}
HadInvalidInst = true;
@@ -712,8 +701,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
if (NeedsBlockEnd) {
// We had some instructions. Early exit
Thread->OpDispatcher->ExitFunction(
Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
break;
}
@@ -724,12 +712,6 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
}
}
#ifdef ZYDIS_DISASSEMBLER
if (FEXCore::Config::Get_X86DISASSEMBLE()) {
LogMan::Msg::IFmt("Guest x86 End");
}
#endif
Thread->OpDispatcher->Finalize();
Thread->FrontendDecoder->DelayedDisownBuffer();
@@ -757,24 +739,37 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
.TotalInstructionsLength = TotalInstructionsLength,
.StartAddr = Thread->FrontendDecoder->DecodedMinAddress,
.Length = Thread->FrontendDecoder->DecodedMaxAddress - Thread->FrontendDecoder->DecodedMinAddress,
.NeedsAddGuestCodeRanges = !HasCustomIR,
};
}
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
// JIT Code object cache lookup
if (CodeObjectCacheService) {
auto CodeCacheEntry = CodeObjectCacheService->FetchCodeObjectFromCache(GuestRIP);
if (CodeCacheEntry) {
auto CompiledCode = Thread->CPUBackend->RelocateJITObjectCode(GuestRIP, CodeCacheEntry);
if (CompiledCode) {
return {
.CompiledCode = CompiledCode,
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
.StartAddr = 0, // Unused
.Length = 0, // Unused
};
}
}
}
if (SourcecodeResolver && Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
if (MappedSection) {
MappedSection->FileInfo.SourcecodeMap =
SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, CodeMap::GetBaseFilename(MappedSection->FileInfo, false));
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (AOTIRCacheEntry.Entry && !AOTIRCacheEntry.Entry->ContainsCode) {
AOTIRCacheEntry.Entry->SourcecodeMap = SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
}
}
// Generate IR + Meta Info
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, NeedsAddGuestCodeRanges] =
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
if (!IRView) {
return {{}, nullptr, 0, 0, false};
return {nullptr, nullptr, 0, 0};
}
// Attempt to get the CPU backend to compile this code
@@ -783,13 +778,9 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
// but this would increase lock contention. Redundant frontend runs aren't
// as expensive and are easily reverted.
if (MaxInst != 1) {
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
Thread->OpDispatcher->DelayedDisownBuffer();
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
.DebugData = nullptr,
.StartAddr = 0,
.Length = 0,
.NeedsAddGuestCodeRanges = false};
return {.CompiledCode = reinterpret_cast<uint8_t*>(Block), .DebugData = nullptr, .StartAddr = 0, .Length = 0};
}
}
@@ -804,11 +795,13 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
Thread->OpDispatcher->DelayedDisownBuffer();
return {
.CompiledCode = std::move(CompiledCode),
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
// In the future with code caching getting wired up, we will pass the rest of the data forward.
// TODO: Pass the data forward when code caching is wired up to this.
.CompiledCode = CompiledCode.BlockEntry,
.DebugData = std::move(DebugData),
.StartAddr = StartAddr,
.Length = Length,
.NeedsAddGuestCodeRanges = NeedsAddGuestCodeRanges,
};
}
@@ -824,15 +817,11 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
// Is the code in the cache?
// The backends only check L1 and L2, not L3
if (auto HostCode = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
if (auto HostCode = Thread->LookupCache->FindBlock(GuestRIP)) {
return HostCode;
}
// Accumulate a JIT count now, as even if another thread raced us, it should count as a compile.
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedJITCount, 1);
auto [CompiledCode, DebugData, StartAddr, Length, NeedsAddGuestCodeRanges] = CompileCode(Thread, GuestRIP, MaxInst);
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
if (CodePtr == nullptr) {
return 0;
} else if (!DebugData) {
@@ -842,75 +831,58 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
// The core managed to compile the code.
if (Config.BlockJITNaming()) {
auto FragmentBasePtr = CompiledCode.BlockBegin;
auto FragmentBasePtr = reinterpret_cast<uint8_t*>(CodePtr);
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
if (DebugData) {
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
}
}
}
} else {
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, DebugData->HostCodeSize);
}
}
}
}
if (Config.LibraryJITNaming() || Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
if (MappedSection) {
if (Config.LibraryJITNaming()) {
Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, MappedSection->FileInfo.Filename);
}
if (Config.GDBSymbols()) {
GDBJITRegister(MappedSection->FileInfo, MappedSection->FileStartVA, GuestRIP, (uintptr_t)CodePtr, *DebugData);
}
}
// Tell the object cache service to serialize the code if enabled
if (CodeObjectCacheService && Config.CacheObjectCodeCompilation == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_READWRITE && DebugData) {
CodeObjectCacheService->AsyncAddSerializationJob(
fextl::make_unique<CodeSerialize::AsyncJobHandler::SerializationJobData>(CodeSerialize::AsyncJobHandler::SerializationJobData {
.GuestRIP = GuestRIP,
.GuestCodeLength = Length,
.GuestCodeHash = 0,
.HostCodeBegin = CodePtr,
.HostCodeLength = DebugData->HostCodeSize,
.HostCodeHash = 0,
.ThreadJobRefCount = &Thread->ObjectCacheRefCounter,
.Relocations = std::move(*DebugData->Relocations),
}));
}
// Clear any relocations that might have been generated
if (!CodeCache.IsGeneratingCache) {
Thread->CPUBackend->ClearRelocations();
}
Thread->CPUBackend->ClearRelocations();
fextl::vector<uint64_t> CodePages;
if (NeedsAddGuestCodeRanges) {
// Track in the guest to host map all entrypoints for all pages the compiled block touches, if any page didn't previously
// contain code, inform the frontend so it can setup SMC detection.
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
CodePages.reserve(BlockInfo->CodePages.size());
CodePages.insert(CodePages.end(), BlockInfo->CodePages.begin(), BlockInfo->CodePages.end());
for (auto CodePage : BlockInfo->CodePages) {
if (Thread->LookupCache->AddBlockExecutableRange(Thread, BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) {
// Early exit
return (uintptr_t)CodePtr;
}
// Insert to lookup cache
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
Thread->LookupCache->AddBlockMapping(Thread, GuestAddr, CodePages, HostAddr);
}
if (CodeMapWriter) {
auto Region = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
if (Region && Region->FileStartVA != 0) {
CodeMapWriter->AppendBlock(*Region, GuestRIP);
}
}
// Pages containing this block are added via AddBlockExecutableRange before each page gets accessed in the frontend
Thread->LookupCache->AddBlockMapping(GuestRIP, CodePtr);
return (uintptr_t)CodePtr;
}
@@ -924,8 +896,7 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
auto [CompiledCode, DebugData, StartAddr, Length, _] = CompileCode(Thread, GuestRIP, 1);
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
if (CodePtr == nullptr) {
return 0;
}
@@ -936,39 +907,46 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
return (uintptr_t)CodePtr;
}
void ContextImpl::InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) {
FEXCORE_PROFILE_SCOPED("InvalidateCodeBuffersCodeRange");
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
auto lk = Thread->LookupCache->AcquireLock();
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
std::scoped_lock lk {CodeBufferListLock};
auto it = CodeBufferList.begin();
while (it != CodeBufferList.end()) {
if (auto Strong = it->lock()) {
Strong->LookupCache->InvalidateRange(Start, Length);
it++;
} else {
it = CodeBufferList.erase(it);
auto lower = Thread->LookupCache->CodePages.lower_bound(Start >> 12);
auto upper = Thread->LookupCache->CodePages.upper_bound((Start + Length - 1) >> 12);
for (auto it = lower; it != upper; it++) {
for (auto Address : it->second) {
ContextImpl::ThreadRemoveCodeEntry(Thread, Address);
}
it->second.clear();
}
}
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
InvalidateGuestThreadCodeRange(Thread, Start, Length);
}
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
if (!Thread) {
return;
}
if (!IsMemoryShared) {
IsMemoryShared = true;
UpdateAtomicTSOEmulationConfig();
if (Config.TSOAutoMigration) {
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
Thread->LookupCache->ClearCache();
}
}
}
void ContextImpl::InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
void ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
"be unique_locked here");
// Ensures now-modified mappings aren't cached as being in their previous non-executable state.
// Accessing FrontendDecoder is safe as the thread's code invalidation mutex must be locked here.
Thread->FrontendDecoder->ResetExecutableRangeCache();
if (Thread->LookupCache->InvalidateCacheRange(Start, Length)) {
FEXCORE_PROFILE_SCOPED("InvalidateCallRet");
// This may cause access violations in the thread on Windows as zeroing is not atomic, this is handled by the frontend
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
}
}
void ContextImpl::ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
static_cast<ContextImpl*>(Frame->Thread->CTX)->SyscallHandler->InvalidateGuestCodeRange(Frame->Thread, GuestRIP, 1);
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
}
std::optional<CustomIRResult>
@@ -977,7 +955,7 @@ ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandl
std::unique_lock lk(CustomIRMutex);
auto InsertedIterator = CustomIRHandlers.emplace(Entrypoint, CustomIRHandlerEntry {Handler, Creator, Data});
auto InsertedIterator = CustomIRHandlers.emplace(Entrypoint, std::tuple(Handler, Creator, Data));
HasCustomIRHandlers = true;
if (!InsertedIterator.second) {
@@ -1001,22 +979,21 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
auto Result = AddCustomIREntrypoint(
Entrypoint,
[this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) {
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0, 0, 0);
auto Block = emit->CreateCodeNode(true, 0);
IRHeader.first->Blocks = emit->WrapNode(Block);
emit->SetCurrentCodeBlock(Block);
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0, 0, 0);
auto Block = emit->CreateCodeNode();
IRHeader.first->Blocks = emit->WrapNode(Block);
emit->SetCurrentCodeBlock(Block);
const auto GPRSize = this->Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
const auto GPRSize = GetGPROpSize();
// Thunk entry-points don't get cached, don't need to be padded.
if (GPRSize == IR::OpSize::i64Bit) {
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
R->Reg = IR::PhysicalRegister(IR::RegClass::GPRFixed, X86State::REG_R11).Raw;
} else {
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
offsetof(Core::CPUState, mm[0][0]));
}
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
if (GPRSize == IR::OpSize::i64Bit) {
IR::Ref R = emit->_StoreRegister(emit->_Constant(Entrypoint), GPRSize);
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
} else {
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->_Constant(Entrypoint)),
offsetof(Core::CPUState, mm[0][0]));
}
emit->_ExitFunction(IR::OpSize::i64Bit, emit->_Constant(GuestThunkEntrypoint));
},
ThunkHandler, (void*)GuestThunkEntrypoint);
@@ -1035,7 +1012,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
void ContextImpl::AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) {
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
ForceTSOValidRanges.Insert(ValidRanges);
ForceTSOInstructions.merge(std::move(Instructions));
ForceTSOInstructions.merge(Instructions);
}
void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
@@ -1045,39 +1022,28 @@ void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
ForceTSOInstructions.erase(ForceTSOInstructions.lower_bound(Address), ForceTSOInstructions.upper_bound(Address + Size));
}
void ContextImpl::MarkMonoBackpatcherBlock(uint64_t BlockEntry) {
MonoBackpatcherBlock.store(BlockEntry, std::memory_order_relaxed);
}
void ContextImpl::RemoveCustomIREntrypoint(FEXCore::Core::InternalThreadState* Thread, uintptr_t Entrypoint) {
void ContextImpl::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
std::scoped_lock lk(CustomIRMutex);
InvalidateGuestCodeRange(nullptr, Entrypoint, 1);
CustomIRHandlers.erase(Entrypoint);
HasCustomIRHandlers = !CustomIRHandlers.empty();
SyscallHandler->InvalidateGuestCodeRange(Thread, Entrypoint, 1);
}
void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint8_t Size, uint64_t Address, uint64_t Value) {
auto Thread = Frame->Thread;
auto CTX = static_cast<ContextImpl*>(Thread->CTX);
{
auto lk = GuardSignalDeferringSection(CTX->CodeInvalidationMutex, Thread);
IR::AOTIRCacheEntry* ContextImpl::LoadAOTIRCacheEntry(const fextl::string& filename) {
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
return rv;
}
if (Size == 8) {
*reinterpret_cast<uint64_t*>(Address) = Value;
} else if (Size == 4) {
*reinterpret_cast<uint32_t*>(Address) = Value;
} else {
ERROR_AND_DIE_FMT("Unexpected write size for backpatcher: {}", Size);
}
}
CTX->SyscallHandler->InvalidateGuestCodeRange(Thread, Address, Size);
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
}
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
}
} // namespace FEXCore::Context
@@ -1,10 +1,10 @@
// SPDX-License-Identifier: MIT
#include "Common/VectorRegType.h"
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/X86HelperGen.h"
#include "Utils/MemberFunctionToPointer.h"
#include <FEXCore/Config/Config.h>
@@ -16,17 +16,14 @@
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <CodeEmitter/Emitter.h>
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#endif
#include <array>
#include <bit>
#include <atomic>
#include <condition_variable>
#include <csignal>
#include <cstring>
#include <signal.h>
namespace FEXCore::CPU {
@@ -34,14 +31,12 @@ static void SleepThread(FEXCore::Context::ContextImpl* CTX, FEXCore::Core::CpuSt
CTX->SyscallHandler->SleepThread(CTX, Frame);
}
constexpr size_t MAX_DISPATCHER_CODE_SIZE = FEXCore::Utils::FEX_PAGE_SIZE * 4;
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 4;
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl* ctx)
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
, CTX {ctx} {
EmitDispatcher();
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(GetBufferBase()), MAX_DISPATCHER_CODE_SIZE);
}
Dispatcher::~Dispatcher() {
@@ -86,17 +81,14 @@ void Dispatcher::EmitDispatcher() {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::rsp, 0);
str(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
ARMEmitter::ForwardLabel CompileSingleStep;
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
FillStaticRegs();
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
ARMEmitter::BiDirectionalLabel LoopTop {};
ARMEmitter::ForwardLabel CompileSingleStep;
#ifdef ARCHITECTURE_arm64ec
(void)b(&LoopTop);
#ifdef _M_ARM_64EC
b(&LoopTop);
AbsoluteLoopTopAddressEnterECFillSRA = GetCursorAddress<uint64_t>();
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
@@ -104,10 +96,10 @@ void Dispatcher::EmitDispatcher() {
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
// Force a single instruction block if ENTRY_FILL_SRA_SINGLE_INST_REG is nonzero entering the JIT, used for inline SMC handling.
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
// Enter JIT
(void)b(&LoopTop);
b(&LoopTop);
AbsoluteLoopTopAddressEnterEC = GetCursorAddress<uint64_t>();
// Load ThreadState and write the target PC there
@@ -119,121 +111,93 @@ void Dispatcher::EmitDispatcher() {
add(ARMEmitter::Size::i64Bit, StaticRegisters[X86State::REG_RSP], ARMEmitter::Reg::rsp, 0);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, TMP1, 0);
ldr(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
FillSpecialRegs(TMP1, TMP2, false, true);
// As ARM64EC uses this as an entrypoint for both guest calls and host returns, opportunistically try to return
// using the call-ret stack to avoid unbalancing it.
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, REG_CALLRET_SP);
// EC_CALL_CHECKER_PC_REG is REG_PF which isn't touched by any of the above
sub(ARMEmitter::Size::i64Bit, TMP1, EC_CALL_CHECKER_PC_REG, TMP1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
// If the entry at the TOS is for the target address, pop it and return to the JIT code
add(ARMEmitter::Size::i64Bit, REG_CALLRET_SP, REG_CALLRET_SP, 0x10);
ret(TMP2);
// Enter JIT
#endif
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
ARMEmitter::BiDirectionalLabel FullLookup {};
ARMEmitter::BiDirectionalLabel CallBlock {};
(void)Bind(&LoopTop);
Bind(&LoopTop);
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
// Load in our RIP
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
#ifdef ARCHITECTURE_arm64ec
// Clobbers TMP1/2
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
ARMEmitter::ForwardLabel l_NotECCode;
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
(void)tbz(TMP1, 0, &l_NotECCode);
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
mov(EC_CALL_CHECKER_PC_REG, RipReg);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionEC));
br(TMP2);
(void)Bind(&l_NotECCode);
#endif
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
// L1 Cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP1, TMP1, 0);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
br(TMP4);
// L1C check failed, do a full lookup
Bind(&FullLookup);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
}
ARMEmitter::ForwardLabel NoBlock;
if (DisableL2Cache()) {
(void)b(&NoBlock);
} else {
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.L2Pointer));
{
// Offset the address and add to our page pointer
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
}
// Load the pointer from the offset
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
// If page pointer is zero then we have no block
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
// Shift the offset by the size of the block cache entry
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
// The the full LookupCacheEntry with a single LDP.
// Check the guest address first to ensure it maps to the address we are currently at.
// This fixes aliasing problems
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
// If the guest address doesn't match, Compile the block.
sub(TMP2, TMP2, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
// Check the host address to see if it matches, else compile the block.
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
// If we've made it here then we have a real compiled block
{
// Offset the address and add to our page pointer
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
// update L1 cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
// Load the pointer from the offset
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
// If page pointer is zero then we have no block
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
// Shift the offset by the size of the block cache entry
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
// The the full LookupCacheEntry with a single LDP.
// Check the guest address first to ensure it maps to the address we are currently at.
// This fixes aliasing problems
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
// If the guest address doesn't match, Compile the block.
sub(TMP2, TMP2, RipReg);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
// Check the host address to see if it matches, else compile the block.
(void)cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
// L1Mask is pre-shifted.
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg.R(), ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
add(TMP1, TMP1, TMP2);
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
// Jump to the block
br(TMP4);
}
// Jump to the block
br(TMP4);
}
}
@@ -258,7 +222,7 @@ void Dispatcher::EmitDispatcher() {
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
#endif
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
@@ -266,7 +230,7 @@ void Dispatcher::EmitDispatcher() {
Body();
#ifdef ARCHITECTURE_arm64ec
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
strb(ARMEmitter::WReg::zr, TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
#endif
@@ -290,7 +254,7 @@ void Dispatcher::EmitDispatcher() {
mov(ARMEmitter::XReg::x0, STATE);
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionLink));
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
} else {
@@ -307,9 +271,37 @@ void Dispatcher::EmitDispatcher() {
br(TMP1);
}
#ifdef _M_ARM_64EC
// Clobbers TMP1/2
auto EmitECExitCheck = [&]() {
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
ARMEmitter::ForwardLabel l_NotECCode;
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
tbz(TMP1, 0, &l_NotECCode);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
mov(EC_CALL_CHECKER_PC_REG, RipReg);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
Bind(&l_NotECCode);
};
#endif
// Need to create the block
{
(void)Bind(&NoBlock);
Bind(&NoBlock);
#ifdef _M_ARM_64EC
EmitECExitCheck();
#endif
EmitSignalGuardedRegion([&]() {
SpillStaticRegs(TMP1);
@@ -343,7 +335,11 @@ void Dispatcher::EmitDispatcher() {
}
{
(void)Bind(&CompileSingleStep);
Bind(&CompileSingleStep);
#ifdef _M_ARM_64EC
EmitECExitCheck();
#endif
EmitSignalGuardedRegion([&]() {
SpillStaticRegs(TMP1);
@@ -487,7 +483,7 @@ void Dispatcher::EmitDispatcher() {
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.ThunkCallbackRet));
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->X86CodeGen.CallbackReturn);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, CTX->Config.Is64BitMode ? 16 : 12);
@@ -502,10 +498,9 @@ void Dispatcher::EmitDispatcher() {
// load static regs
FillStaticRegs();
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
// Now go back to the regular dispatcher loop
(void)b(&LoopTop);
b(&LoopTop);
}
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
@@ -543,8 +538,8 @@ void Dispatcher::EmitDispatcher() {
return Address;
};
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.LUDIV));
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.LDIV));
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
// Interpreter fallbacks
{
@@ -555,8 +550,8 @@ void Dispatcher::EmitDispatcher() {
FABI_F80_I16_I32_PTR,
FABI_F32_I16_F80_PTR,
FABI_F64_I16_F80_PTR,
FABI_F64_F64_PTR,
FABI_F64_F64_F64_PTR,
FABI_F64_I16_F64_PTR,
FABI_F64_I16_F64_F64_PTR,
FABI_I16_I16_F80_PTR,
FABI_I32_I16_F80_PTR,
FABI_I64_I16_F80_PTR,
@@ -564,7 +559,7 @@ void Dispatcher::EmitDispatcher() {
FABI_F80_I16_F80_PTR,
FABI_F80_I16_F80_F80_PTR,
FABI_F80x2_I16_F80_PTR,
FABI_F64x2_F64_PTR,
FABI_F64x2_I16_F64_PTR,
FABI_I32_I64_I64_V128_V128_I16,
FABI_I32_V128_V128_I16,
}};
@@ -574,15 +569,14 @@ void Dispatcher::EmitDispatcher() {
}
}
(void)Bind(&l_CTX);
Bind(&l_CTX);
dc64(reinterpret_cast<uintptr_t>(CTX));
(void)Bind(&l_Sleep);
Bind(&l_Sleep);
dc64(reinterpret_cast<uint64_t>(SleepThread));
(void)Bind(&l_CompileBlock);
Bind(&l_CompileBlock);
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
dc64(PMFCompileBlock.GetConvertedPointer());
(void)Bind(&l_CompileSingleStep);
Bind(&l_CompileSingleStep);
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
dc64(PMFCompileSingleStep.GetConvertedPointer());
@@ -613,7 +607,6 @@ void Dispatcher::EmitDispatcher() {
#ifdef VIXL_SIMULATOR
void Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.WriteXRegister(1, 0);
Simulator.RunFrom(reinterpret_cast< const vixl::aarch64::Instruction*>(DispatchPtr));
}
@@ -625,226 +618,6 @@ void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_
#endif
void Dispatcher::EmitI32ToExtF80() {
ARMEmitter::ForwardLabel ZeroCase;
ARMEmitter::ForwardLabel Done;
(void)cbz(ARMEmitter::Size::i32Bit, TMP2, &ZeroCase);
lsr(ARMEmitter::Size::i32Bit, TMP4, TMP2, 31);
tst(ARMEmitter::Size::i32Bit, TMP2, TMP2);
neg(ARMEmitter::Size::i32Bit, TMP3, TMP2);
csel(ARMEmitter::Size::i32Bit, TMP3, TMP3, TMP2, ARMEmitter::Condition::CC_MI);
clz(ARMEmitter::Size::i32Bit, TMP1, TMP3);
mov(ARMEmitter::Size::i32Bit, TMP2, 0x401E);
sub(ARMEmitter::Size::i32Bit, TMP2, TMP2, TMP1);
orr(ARMEmitter::Size::i32Bit, TMP2, TMP2, TMP4, ARMEmitter::ShiftType::LSL, 15);
lslv(ARMEmitter::Size::i32Bit, TMP3, TMP3, TMP1);
lsl(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)b(&Done);
(void)Bind(&ZeroCase);
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
(void)Bind(&Done);
}
void Dispatcher::EmitI16ToExtF80() {
sxth(ARMEmitter::Size::i32Bit, TMP2, TMP2);
ARMEmitter::ForwardLabel ZeroCase;
ARMEmitter::ForwardLabel Done;
(void)cbz(ARMEmitter::Size::i32Bit, TMP2, &ZeroCase);
lsr(ARMEmitter::Size::i32Bit, TMP4, TMP2, 31);
tst(ARMEmitter::Size::i32Bit, TMP2, TMP2);
neg(ARMEmitter::Size::i32Bit, TMP3, TMP2);
csel(ARMEmitter::Size::i32Bit, TMP3, TMP3, TMP2, ARMEmitter::Condition::CC_MI);
clz(ARMEmitter::Size::i32Bit, TMP1, TMP3);
mov(ARMEmitter::Size::i32Bit, TMP2, 0x401E);
sub(ARMEmitter::Size::i32Bit, TMP2, TMP2, TMP1);
orr(ARMEmitter::Size::i32Bit, TMP2, TMP2, TMP4, ARMEmitter::ShiftType::LSL, 15);
lslv(ARMEmitter::Size::i32Bit, TMP3, TMP3, TMP1);
lsl(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)b(&Done);
(void)Bind(&ZeroCase);
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
(void)Bind(&Done);
}
void Dispatcher::EmitF32ToExtF80() {
ARMEmitter::ForwardLabel InfNaN;
ARMEmitter::ForwardLabel ZeroDenormal;
ARMEmitter::ForwardLabel Denormal;
ARMEmitter::ForwardLabel NaN;
ARMEmitter::ForwardLabel Done;
ARMEmitter::BiDirectionalLabel NormalPath;
ARMEmitter::ForwardLabel ZeroResult;
fmov(ARMEmitter::Size::i32Bit, TMP1, VTMP1.S());
ubfx(ARMEmitter::Size::i32Bit, TMP2, TMP1, 23, 8);
and_(ARMEmitter::Size::i32Bit, TMP3, TMP1, 0x007FFFFF);
lsr(ARMEmitter::Size::i32Bit, TMP4, TMP1, 31);
cmp(ARMEmitter::Size::i32Bit, TMP2, 0xFF);
(void)b(ARMEmitter::Condition::CC_EQ, &InfNaN);
(void)cbz(ARMEmitter::Size::i32Bit, TMP2, &ZeroDenormal);
(void)Bind(&NormalPath);
// Exponent bias adjustment, where bias is 0x3F80
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0x3F80);
add(ARMEmitter::Size::i32Bit, TMP2, TMP2, TMP1);
orr(ARMEmitter::Size::i32Bit, TMP2, TMP2, TMP4, ARMEmitter::ShiftType::LSL, 15);
// Set implicit bit and shift fraction to extF80 position
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, 0x00800000ULL);
orr(ARMEmitter::Size::i64Bit, TMP3, TMP3, TMP1);
lsl(ARMEmitter::Size::i64Bit, TMP3, TMP3, 40);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)b(&Done);
(void)Bind(&ZeroDenormal);
(void)cbz(ARMEmitter::Size::i32Bit, TMP3, &ZeroResult);
(void)Bind(&Denormal);
clz(ARMEmitter::Size::i32Bit, TMP1, TMP3);
sub(ARMEmitter::Size::i32Bit, TMP1, TMP1, 8);
mov(ARMEmitter::Size::i32Bit, TMP2, 1);
sub(ARMEmitter::Size::i32Bit, TMP2, TMP2, TMP1);
lslv(ARMEmitter::Size::i32Bit, TMP3, TMP3, TMP1);
(void)b(&NormalPath);
(void)Bind(&ZeroResult);
lsl(ARMEmitter::Size::i32Bit, TMP2, TMP4, 15);
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)b(&Done);
(void)Bind(&InfNaN);
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &NaN);
lsl(ARMEmitter::Size::i32Bit, TMP2, TMP4, 15);
orr(ARMEmitter::Size::i32Bit, TMP2, TMP2, 0x7FFF);
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, 0x8000000000000000ULL);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)b(&Done);
(void)Bind(&NaN);
lsl(ARMEmitter::Size::i32Bit, TMP2, TMP4, 15);
orr(ARMEmitter::Size::i32Bit, TMP2, TMP2, 0x7FFF);
lsl(ARMEmitter::Size::i64Bit, TMP3, TMP3, 40);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, 0xC000000000000000ULL);
orr(ARMEmitter::Size::i64Bit, TMP3, TMP3, TMP1);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)Bind(&Done);
}
void Dispatcher::EmitF64ToExtF80() {
ARMEmitter::ForwardLabel InfNaN;
ARMEmitter::ForwardLabel ZeroDenormal;
ARMEmitter::ForwardLabel Denormal;
ARMEmitter::ForwardLabel NaN;
ARMEmitter::ForwardLabel Done;
ARMEmitter::BiDirectionalLabel NormalPath;
ARMEmitter::ForwardLabel ZeroResult;
fmov(ARMEmitter::Size::i64Bit, TMP1, VTMP1.D());
lsr(ARMEmitter::Size::i64Bit, TMP4, TMP1, 63);
ubfx(ARMEmitter::Size::i64Bit, TMP2, TMP1, 52, 11);
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, 0x000FFFFFFFFFFFFFULL);
and_(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP3);
cmp(ARMEmitter::Size::i64Bit, TMP2, 0x7FF);
(void)b(ARMEmitter::Condition::CC_EQ, &InfNaN);
(void)cbz(ARMEmitter::Size::i64Bit, TMP2, &ZeroDenormal);
(void)Bind(&NormalPath);
// Exponent bias adjustment where bias difference is 0x3C00
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x3000);
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0xC00);
orr(ARMEmitter::Size::i64Bit, TMP2, TMP2, TMP4, ARMEmitter::ShiftType::LSL, 15);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, 0x0010000000000000ULL);
orr(ARMEmitter::Size::i64Bit, TMP3, TMP3, TMP1);
lsl(ARMEmitter::Size::i64Bit, TMP3, TMP3, 11);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)b(&Done);
(void)Bind(&ZeroDenormal);
(void)cbz(ARMEmitter::Size::i64Bit, TMP3, &ZeroResult);
(void)Bind(&Denormal);
clz(ARMEmitter::Size::i64Bit, TMP1, TMP3);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 11);
mov(ARMEmitter::Size::i64Bit, TMP2, 1);
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, TMP1);
lslv(ARMEmitter::Size::i64Bit, TMP3, TMP3, TMP1);
(void)b(&NormalPath);
(void)Bind(&ZeroResult);
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP4, 15);
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)b(&Done);
(void)Bind(&InfNaN);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP3, &NaN);
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP4, 15);
orr(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x7FFF);
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, 0x8000000000000000ULL);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)b(&Done);
(void)Bind(&NaN);
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP4, 15);
orr(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x7FFF);
lsl(ARMEmitter::Size::i64Bit, TMP3, TMP3, 11);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, 0xC000000000000000ULL);
orr(ARMEmitter::Size::i64Bit, TMP3, TMP3, TMP1);
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 4, TMP2);
(void)Bind(&Done);
}
uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
auto Address = GetCursorAddress<uint64_t>();
constexpr static auto FallbackPointerReg = TMP4;
@@ -915,33 +688,65 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
switch (ABI) {
case FABI_F80_I16_F32_PTR: {
// Save NZCV - it's a static register (guest x86 flags) and the inline code clobbers it
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
EmitF32ToExtF80();
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
// vtmp1 (v0/v16): source
SpillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, TMP3, true);
if (!TMP_ABIARGS) {
fmov(VABI1.S(), VTMP1.S());
}
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
mov(ARMEmitter::XReg::x1, STATE);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, float, uint64_t>(FallbackPointerReg);
} else {
blr(FallbackPointerReg);
}
FillF80Result();
} break;
case FABI_F80_I16_F64_PTR: {
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
EmitF64ToExtF80();
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
} break;
case FABI_F80_I16_I16_PTR: {
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
EmitI16ToExtF80();
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
// vtmp1 (v0/v16): source
SpillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, TMP3, true);
if (!TMP_ABIARGS) {
fmov(VABI1.D(), VTMP1.D());
}
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
mov(ARMEmitter::XReg::x1, STATE);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, double, uint64_t>(FallbackPointerReg);
} else {
blr(FallbackPointerReg);
}
FillF80Result();
} break;
case FABI_F80_I16_I16_PTR:
case FABI_F80_I16_I32_PTR: {
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
EmitI32ToExtF80();
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
// tmp2 (x1/x11): source
SpillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, TMP3, true);
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, TMP2);
}
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
mov(ARMEmitter::XReg::x2, STATE);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, uint32_t, uint64_t>(FallbackPointerReg);
} else {
blr(FallbackPointerReg);
}
FillF80Result();
} break;
case FABI_F32_I16_F80_PTR: {
// Linux Reg/Win32 Reg:
@@ -952,7 +757,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
if (!TMP_ABIARGS) {
mov(VABI1.Q(), VTMP1.Q());
fmov(VABI1.D(), VTMP1.D());
}
mov(ARMEmitter::XReg::x1, STATE);
@@ -985,7 +790,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
FillF64Result();
} break;
case FABI_F64_F64_PTR: {
case FABI_F64_I16_F64_PTR: {
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
@@ -995,17 +800,18 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
if (!TMP_ABIARGS) {
fmov(VABI1.D(), VTMP1.D());
}
mov(ARMEmitter::XReg::x0, STATE);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
mov(ARMEmitter::XReg::x1, STATE);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<double, double, uint64_t>(FallbackPointerReg);
GenerateIndirectRuntimeCall<double, uint16_t, double, uint64_t>(FallbackPointerReg);
} else {
blr(FallbackPointerReg);
}
FillF64Result();
} break;
case FABI_F64_F64_F64_PTR: {
case FABI_F64_I16_F64_F64_PTR: {
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
@@ -1018,9 +824,10 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
fmov(VABI2.D(), VTMP2.D());
}
mov(ARMEmitter::XReg::x0, STATE);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
mov(ARMEmitter::XReg::x1, STATE);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<double, double, double, uint64_t>(FallbackPointerReg);
GenerateIndirectRuntimeCall<double, uint16_t, double, double, uint64_t>(FallbackPointerReg);
} else {
blr(FallbackPointerReg);
}
@@ -1180,7 +987,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
FillF80x2Result();
} break;
case FABI_F64x2_F64_PTR: {
case FABI_F64x2_I16_F64_PTR: {
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
@@ -1189,13 +996,14 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
SpillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, TMP3, true);
mov(ARMEmitter::XReg::x0, STATE);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
mov(ARMEmitter::XReg::x1, STATE);
if (!TMP_ABIARGS) {
fmov(VABI1.D(), VTMP1.D());
}
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
// GenerateIndirectRuntimeCall<FEXCore::VectorScalarF64Pair, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
// GenerateIndirectRuntimeCall<FEXCore::VectorScalarF64Pair, uint16_t, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
} else {
blr(FallbackPointerReg);
}
@@ -1273,75 +1081,30 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto& Ptrs = Thread->CurrentFrame->Pointers;
auto& Common = Thread->CurrentFrame->Pointers.Common;
Ptrs.DispatcherLoopTop = AbsoluteLoopTopAddress;
Ptrs.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Ptrs.DispatcherLoopTopEnterEC = AbsoluteLoopTopAddressEnterEC;
Ptrs.DispatcherLoopTopEnterECFillSRA = AbsoluteLoopTopAddressEnterECFillSRA;
Ptrs.ExitFunctionLinker = ExitFunctionLinkerAddress;
Ptrs.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
Ptrs.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
Ptrs.GuestSignal_SIGILL = GuestSignal_SIGILL;
Ptrs.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
Ptrs.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
Ptrs.SignalReturnHandler = SignalHandlerReturnAddress;
Ptrs.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
Ptrs.LUDIVHandler = LUDIVHandlerAddress;
Ptrs.LDIVHandler = LDIVHandlerAddress;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Common.DispatcherLoopTopEnterEC = AbsoluteLoopTopAddressEnterEC;
Common.DispatcherLoopTopEnterECFillSRA = AbsoluteLoopTopAddressEnterECFillSRA;
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
Common.SignalReturnHandler = SignalHandlerReturnAddress;
Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
auto& AArch64 = Thread->CurrentFrame->Pointers.AArch64;
AArch64.LUDIVHandler = LUDIVHandlerAddress;
AArch64.LDIVHandler = LDIVHandlerAddress;
// Fill in the fallback handlers
InterpreterOps::FillFallbackIndexPointers(Ptrs.FallbackHandlerPointers, &ABIPointers[0]);
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers, &ABIPointers[0]);
}
}
SignalDelegatorConfig Dispatcher::MakeSignalDelegatorConfig() const {
// PF/AF are the final two SRA registers. We only want GPRs
const auto GPRCount = uint16_t(StaticRegisters.size() - 2);
const auto FPRCount = uint16_t(StaticFPRegisters.size());
const auto GetSRAGPRMapping = [GPRCount, this] {
SignalDelegatorConfig::SRAIndexMapping Mapping {};
for (size_t i = 0; i < GPRCount; ++i) {
Mapping[i] = StaticRegisters[i].Idx();
}
return Mapping;
};
const auto GetSRAFPRMapping = [FPRCount, this] {
SignalDelegatorConfig::SRAIndexMapping Mapping {};
for (size_t i = 0; i < FPRCount; ++i) {
Mapping[i] = StaticFPRegisters[i].Idx();
}
return Mapping;
};
return FEXCore::SignalDelegatorConfig {
.DispatcherBegin = Start,
.DispatcherEnd = End,
.AbsoluteLoopTopAddress = AbsoluteLoopTopAddress,
.AbsoluteLoopTopAddressFillSRA = AbsoluteLoopTopAddressFillSRA,
.SignalHandlerReturnAddress = SignalHandlerReturnAddress,
.SignalHandlerReturnAddressRT = SignalHandlerReturnAddressRT,
.PauseReturnInstruction = PauseReturnInstruction,
.ThreadPauseHandlerAddressSpillSRA = ThreadPauseHandlerAddressSpillSRA,
.ThreadPauseHandlerAddress = ThreadPauseHandlerAddress,
// Stop handlers.
.ThreadStopHandlerAddressSpillSRA = ThreadStopHandlerAddressSpillSRA,
.ThreadStopHandlerAddress = ThreadStopHandlerAddress,
// SRA information.
.SRAGPRCount = GPRCount,
.SRAFPRCount = FPRCount,
.SRAGPRMapping = GetSRAGPRMapping(),
.SRAFPRMapping = GetSRAFPRMapping(),
};
}
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl* CTX) {
return fextl::make_unique<Dispatcher>(CTX);
}
@@ -2,19 +2,25 @@
#pragma once
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/fextl/memory.h>
#include <array>
#include <cstddef>
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#endif
#include <cstdint>
#include <signal.h>
#include <stddef.h>
#include <stack>
#include <tuple>
namespace FEXCore {
struct GuestSigAction;
struct SignalDelegatorConfig;
} // namespace FEXCore
}
namespace FEXCore::Core {
struct CpuStateFrame;
@@ -36,36 +42,6 @@ public:
Dispatcher(FEXCore::Context::ContextImpl* ctx);
~Dispatcher();
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
#ifdef VIXL_SIMULATOR
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
#else
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
DispatchPtr(Frame, false);
}
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
CallbackPtr(Frame, RIP);
}
#endif
uint64_t GetExitFunctionLinkerAddress() const {
return ExitFunctionLinkerAddress;
}
SignalDelegatorConfig MakeSignalDelegatorConfig() const;
protected:
FEXCore::Context::ContextImpl* CTX;
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame, bool SingleInst);
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
AsmDispatch DispatchPtr;
JITCallback CallbackPtr;
private:
/**
* @name Dispatch Helper functions
* @{ */
@@ -83,29 +59,68 @@ private:
uint64_t GuestSignal_SIGILL {};
uint64_t GuestSignal_SIGTRAP {};
uint64_t GuestSignal_SIGSEGV {};
uint64_t IntCallbackReturnAddress {};
uint64_t PauseReturnInstruction {};
std::array<uint64_t, FallbackABI::FABI_UNKNOWN> ABIPointers {};
/** @} */
uint64_t Start {};
uint64_t End {};
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
#ifdef VIXL_SIMULATOR
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
#else
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
DispatchPtr(Frame);
}
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
CallbackPtr(Frame, RIP);
}
#endif
uint16_t GetSRAGPRCount() const {
// PF/AF are the final two SRA registers.
// Only return the SRA for GPRs.
return StaticRegisters.size() - 2;
}
uint16_t GetSRAFPRCount() const {
return StaticFPRegisters.size();
}
void GetSRAGPRMapping(uint8_t Mapping[16]) const {
for (size_t i = 0; i < StaticRegisters.size() - 2; ++i) {
Mapping[i] = StaticRegisters[i].Idx();
}
}
void GetSRAFPRMapping(uint8_t Mapping[16]) const {
for (size_t i = 0; i < StaticFPRegisters.size(); ++i) {
Mapping[i] = StaticFPRegisters[i].Idx();
}
}
protected:
FEXCore::Context::ContextImpl* CTX;
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame);
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
AsmDispatch DispatchPtr;
JITCallback CallbackPtr;
private:
// Long division helpers
uint64_t LUDIVHandlerAddress {};
uint64_t LDIVHandlerAddress {};
void EmitDispatcher();
uint64_t GenerateABICall(FallbackABI ABI);
// Inline softfloat conversion emitters - avoid FPCR save/restore overhead
// These emit ARM64 code that performs the conversion using only integer ops
void EmitI16ToExtF80();
void EmitI32ToExtF80();
void EmitF32ToExtF80();
void EmitF64ToExtF80();
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
};
} // namespace FEXCore::CPU
File diff suppressed because it is too large. Load diff
+16 -61
View File
@@ -2,59 +2,40 @@
#pragma once
#include "Interface/Core/X86Tables/X86Tables.h"
#include "Interface/IR/IR.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CodeCache.h>
#include <FEXCore/Utils/ThreadPoolAllocator.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/vector.h>
#include <FEXCore/fextl/robin_map.h>
#include <array>
#include <cstddef>
#include <cstdint>
#include <optional>
#include <stddef.h>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::HLE {
enum class SyscallOSABI;
}
namespace FEXCore::Frontend {
class Decoder final {
public:
enum class DecodedBlockStatus {
SUCCESS,
INVALID_INST,
NOEXEC_INST,
PARTIAL_DECODE_INST,
BAD_RELOCATION,
};
// New Frontend decoding
struct DecodedBlocks final {
uint64_t Entry {};
uint64_t Size {};
uint64_t NumInstructions {};
FEXCore::X86Tables::DecodedInst* DecodedInstructions;
DecodedBlockStatus BlockStatus;
bool IsEntryPoint {};
bool ForceFullSMCDetection {};
bool HasInvalidInstruction {};
};
struct DecodedBlockInformation final {
uint64_t TotalInstructionCount;
bool Is64BitMode {};
fextl::vector<DecodedBlocks> Blocks;
fextl::set<uint64_t> EntryPoints;
fextl::set<uint64_t> CodePages; // Start addresses of all pages touching the block
};
Decoder(FEXCore::Core::InternalThreadState* Thread);
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
Decoder(FEXCore::Context::ContextImpl* ctx);
void DecodeInstructionsAtEntry(const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst,
std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
const DecodedBlockInformation* GetDecodedBlockInfo() const {
return &BlockInfo;
@@ -63,6 +44,9 @@ public:
uint64_t DecodedMinAddress {};
uint64_t DecodedMaxAddress {~0ULL};
void SetSectionMaxAddress(uint64_t v) {
SectionMaxAddress = v;
}
void SetExternalBranches(fextl::set<uint64_t>* v) {
ExternalBranches = v;
}
@@ -71,10 +55,6 @@ public:
PoolObject.DelayedDisownBuffer();
}
void ResetExecutableRangeCache() {
ExecutableRangeBase = ExecutableRangeEnd = 0;
}
private:
// To pass any information from instruction prefixes
// down into the actual instruction handling machinery.
@@ -84,27 +64,19 @@ private:
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
};
FEXCore::Core::InternalThreadState* Thread;
FEXCore::Context::ContextImpl* CTX;
const FEXCore::HLE::SyscallOSABI OSABI {};
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
bool DecodeInstructionImpl(uint64_t PC);
DecodedBlockStatus DecodeInstruction(uint64_t PC);
bool DecodeInstruction(uint64_t PC);
void BranchTargetInMultiblockRange();
bool IsBranchMonoTailcall(uint64_t NumInstructions) const;
bool InstCanContinue() const;
void AddBranchTarget(uint64_t Target);
bool CheckRangeExecutable(uint64_t Address, uint64_t Size);
uint8_t ReadByte();
std::optional<uint8_t> PeekByte(uint8_t Offset);
std::pair<uint64_t, bool> ReadData(uint8_t Size);
uint8_t PeekByte(uint8_t Offset) const;
uint64_t ReadData(uint8_t Size);
void SkipBytes(uint8_t Size) {
InstructionSize += Size;
}
@@ -112,36 +84,26 @@ private:
bool NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
void DecodeREXIfValid(int8_t ExpectedOffset = -1);
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
Utils::PoolBufferWithTimedRetirement<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
size_t DecodedSize {};
uint64_t ExecutableRangeBase {};
uint64_t ExecutableRangeEnd {};
bool ExecutableRangeWritable {};
bool HitNonExecutableRange {};
bool HitBadRelocation {};
const uint8_t* InstStream {};
IR::OpSize GetGPROpSize() const {
return BlockInfo.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
}
static constexpr size_t MAX_INST_SIZE = 15;
uint8_t InstructionSize {};
std::array<uint8_t, MAX_INST_SIZE> Instruction;
uint8_t LastEscapePrefix {};
FEXCore::X86Tables::DecodedInst* DecodeInst;
// This is for multiblock data tracking
bool SymbolAvailable {false};
uint64_t EntryPoint {};
uint64_t MaxCondBranchForward {};
uint64_t MaxCondBranchBackwards {~0ULL};
uint64_t SymbolMaxAddress {};
uint64_t SymbolMinAddress {~0ULL};
uint64_t SectionMaxAddress {~0ULL};
uint64_t SectionMinAddress {};
uint64_t NextBlockStartAddress {~0ULL};
DecodedBlockInformation BlockInfo;
@@ -150,8 +112,6 @@ private:
fextl::set<uint64_t> VisitedBlocks;
fextl::set<uint64_t>* ExternalBranches {nullptr};
const fextl::robin_map<uint32_t, GuestRelocationType>* Relocations {nullptr};
// ModRM rm decoding
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
@@ -162,11 +122,6 @@ private:
&FEXCore::Frontend::Decoder::DecodeModRM_16,
};
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_X87_TABLE_SIZE>* X87Table;
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_TABLE_SIZE>* VEXTable {};
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_GROUP_TABLE_SIZE>* VEXTableGroup {};
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
};
} // namespace FEXCore::Frontend
@@ -6,7 +6,7 @@
#include "Interface/IR/IR.h"
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/SHMStats.h>
#include <FEXCore/Utils/Profiler.h>
namespace FEXCore::CPU {
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
@@ -36,48 +36,18 @@ FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t
return State;
}
FEXCORE_PRESERVE_ALL_ATTR static void HandleX87Exception(const softfloat_state& State, FEXCore::Core::CpuStateFrame* Frame) {
// Check for Invalid Operation exception (bit 0 of X87 status word)
if (State.exceptionFlags & softfloat_flag_invalid) {
Frame->State.flags[FEXCore::X86State::X87FLAG_IE_LOC] = 1;
}
}
// Wrapper for SoftFloat state to handle X87 exceptions
class ScopedSoftFloatState {
public:
FEXCORE_PRESERVE_ALL_ATTR ScopedSoftFloatState(uint16_t FCW, FEXCore::Core::CpuStateFrame* Frame, bool Force80BitPrecision = false)
: State(SoftFloatStateFromFCW(FCW, Force80BitPrecision))
, Frame(Frame) {}
FEXCORE_PRESERVE_ALL_ATTR ~ScopedSoftFloatState() {
HandleX87Exception(State, Frame);
}
// Disable copy and move to ensure RAII semantics
ScopedSoftFloatState(const ScopedSoftFloatState&) = delete;
ScopedSoftFloatState& operator=(const ScopedSoftFloatState&) = delete;
ScopedSoftFloatState(ScopedSoftFloatState&&) = delete;
ScopedSoftFloatState& operator=(ScopedSoftFloatState&&) = delete;
softfloat_state State;
private:
FEXCore::Core::CpuStateFrame* Frame;
};
template<>
struct OpHandlers<IR::OP_F80CVTTO> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle4(uint16_t FCW, float src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(&State.State, src);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(&State, src);
}
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle8(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(&State.State, src);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(&State, src);
}
};
@@ -85,12 +55,12 @@ template<>
struct OpHandlers<IR::OP_F80CMP> {
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
softfloat_state State = SoftFloatStateFromFCW(FCW);
bool eq, lt, nan;
uint64_t ResultFlags = 0;
X80SoftFloat::FCMP(&State.State, Src1, Src2, &eq, &lt, &nan);
X80SoftFloat::FCMP(&State, Src1, Src2, &eq, &lt, &nan);
if (lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
@@ -108,14 +78,14 @@ template<>
struct OpHandlers<IR::OP_F80CVT> {
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToF32(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToF32(&State);
}
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToF64(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToF64(&State);
}
};
@@ -123,26 +93,26 @@ template<>
struct OpHandlers<IR::OP_F80CVTINT> {
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToI16(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToI16(&State);
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToI32(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToI32(&State);
}
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToI64(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToI64(&State);
}
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
auto rv = extF80_to_i32(&State.State, X80SoftFloat(src), softfloat_round_minMag, false);
softfloat_state State = SoftFloatStateFromFCW(FCW);
auto rv = extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
if (rv > INT16_MAX || rv < INT16_MIN) {
///< Indefinite value for 16-bit conversions.
@@ -154,14 +124,14 @@ struct OpHandlers<IR::OP_F80CVTINT> {
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return extF80_to_i32(&State.State, X80SoftFloat(src), softfloat_round_minMag, false);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
}
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return extF80_to_i64(&State.State, X80SoftFloat(src), softfloat_round_minMag, false);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return extF80_to_i64(&State, X80SoftFloat(src), softfloat_round_minMag, false);
}
};
@@ -182,8 +152,8 @@ template<>
struct OpHandlers<IR::OP_F80ROUND> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FRNDINT(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FRNDINT(&State, Src1);
}
};
@@ -191,8 +161,8 @@ template<>
struct OpHandlers<IR::OP_F80F2XM1> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::F2XM1(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::F2XM1(&State, Src1);
}
};
@@ -200,8 +170,8 @@ template<>
struct OpHandlers<IR::OP_F80TAN> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FTAN(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FTAN(&State, Src1);
}
};
@@ -209,8 +179,8 @@ template<>
struct OpHandlers<IR::OP_F80SQRT> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FSQRT(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FSQRT(&State, Src1);
}
};
@@ -218,8 +188,8 @@ template<>
struct OpHandlers<IR::OP_F80SIN> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FSIN(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FSIN(&State, Src1);
}
};
@@ -227,8 +197,8 @@ template<>
struct OpHandlers<IR::OP_F80COS> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FCOS(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FCOS(&State, Src1);
}
};
@@ -236,8 +206,8 @@ template<>
struct OpHandlers<IR::OP_F80SINCOS> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegPairType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return FEXCore::MakeVectorRegPair(X80SoftFloat::FSIN(&State.State, Src1), X80SoftFloat::FCOS(&State.State, Src1));
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return FEXCore::MakeVectorRegPair(X80SoftFloat::FSIN(&State, Src1), X80SoftFloat::FCOS(&State, Src1));
}
};
@@ -261,8 +231,8 @@ template<>
struct OpHandlers<IR::OP_F80ADD> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FADD(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FADD(&State, Src1, Src2);
}
};
@@ -270,8 +240,8 @@ template<>
struct OpHandlers<IR::OP_F80SUB> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FSUB(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FSUB(&State, Src1, Src2);
}
};
@@ -279,8 +249,8 @@ template<>
struct OpHandlers<IR::OP_F80MUL> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FMUL(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FMUL(&State, Src1, Src2);
}
};
@@ -288,8 +258,8 @@ template<>
struct OpHandlers<IR::OP_F80DIV> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FDIV(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FDIV(&State, Src1, Src2);
}
};
@@ -297,8 +267,8 @@ template<>
struct OpHandlers<IR::OP_F80FYL2X> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FYL2X(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FYL2X(&State, Src1, Src2);
}
};
@@ -306,8 +276,8 @@ template<>
struct OpHandlers<IR::OP_F80ATAN> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FATAN(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FATAN(&State, Src1, Src2);
}
};
@@ -315,8 +285,8 @@ template<>
struct OpHandlers<IR::OP_F80FPREM1> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FREM1(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FREM1(&State, Src1, Src2);
}
};
@@ -324,8 +294,8 @@ template<>
struct OpHandlers<IR::OP_F80FPREM> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FREM(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FREM(&State, Src1, Src2);
}
};
@@ -333,14 +303,14 @@ template<>
struct OpHandlers<IR::OP_F80SCALE> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FSCALE(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FSCALE(&State, Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F64SIN> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return sin(src);
}
@@ -348,7 +318,7 @@ struct OpHandlers<IR::OP_F64SIN> {
template<>
struct OpHandlers<IR::OP_F64COS> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return cos(src);
}
@@ -356,7 +326,7 @@ struct OpHandlers<IR::OP_F64COS> {
template<>
struct OpHandlers<IR::OP_F64SINCOS> {
FEXCORE_PRESERVE_ALL_ATTR static VectorScalarF64Pair handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static VectorScalarF64Pair handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
double sin, cos;
#ifdef _WIN32
@@ -371,7 +341,7 @@ struct OpHandlers<IR::OP_F64SINCOS> {
template<>
struct OpHandlers<IR::OP_F64TAN> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return tan(src);
}
@@ -379,7 +349,7 @@ struct OpHandlers<IR::OP_F64TAN> {
template<>
struct OpHandlers<IR::OP_F64F2XM1> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return exp2(src) - 1.0;
}
@@ -387,7 +357,7 @@ struct OpHandlers<IR::OP_F64F2XM1> {
template<>
struct OpHandlers<IR::OP_F64ATAN> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return atan2(src1, src2);
}
@@ -395,7 +365,7 @@ struct OpHandlers<IR::OP_F64ATAN> {
template<>
struct OpHandlers<IR::OP_F64FPREM> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return fmod(src1, src2);
}
@@ -403,7 +373,7 @@ struct OpHandlers<IR::OP_F64FPREM> {
template<>
struct OpHandlers<IR::OP_F64FPREM1> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return remainder(src1, src2);
}
@@ -411,7 +381,7 @@ struct OpHandlers<IR::OP_F64FPREM1> {
template<>
struct OpHandlers<IR::OP_F64FYL2X> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return src2 * log2(src1);
}
@@ -419,7 +389,7 @@ struct OpHandlers<IR::OP_F64FYL2X> {
template<>
struct OpHandlers<IR::OP_F64SCALE> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
if (src1 == 0.0) { // src1 might be +/- zero
return src1; // this will return negative or positive zero if when appropriate
@@ -434,17 +404,18 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1q, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
X80SoftFloat Src1 = Src1q;
ScopedSoftFloatState State {FCW, Frame};
bool Negative = Src1.Top.Sign;
softfloat_state State = SoftFloatStateFromFCW(FCW);
bool Negative = Src1.Sign;
Src1 = X80SoftFloat::FRNDINT(&State.State, Src1);
Src1 = X80SoftFloat::FRNDINT(&State, Src1);
// Clear the Sign bit
Src1.Top.Sign = 0;
Src1.Sign = 0;
uint64_t Tmp = Src1.ToI64(&State.State);
uint64_t Tmp = Src1.ToI64(&State);
X80SoftFloat Rv;
uint8_t* BCD = reinterpret_cast<uint8_t*>(&Rv);
memset(BCD, 0, 10);
for (size_t i = 0; i < 9; ++i) {
if (Tmp == 0) {
@@ -503,7 +474,7 @@ struct OpHandlers<IR::OP_F80BCDLOAD> {
X80SoftFloat Tmp;
Tmp = BCD;
Tmp.Top.Sign = Negative;
Tmp.Sign = Negative;
return Tmp;
}
};
@@ -82,22 +82,24 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle)};
// Double Precision Unary
Info[Core::OPINDEX_F64SIN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle)};
Info[Core::OPINDEX_F64COS] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle)};
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_F64_PTR],
Info[Core::OPINDEX_F64SIN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle)};
Info[Core::OPINDEX_F64COS] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle)};
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_I16_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SINCOS>::handle)};
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_I16_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
// Double Precision Binary
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_F64_F64_PTR],
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle)};
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle)};
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_F64_F64_PTR],
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle)};
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_F64_F64_PTR],
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle)};
// SSE4.2 string instructions
@@ -218,21 +220,21 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
return true; \
}
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
}
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64x2_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64x2_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
}
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_I16_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
}
// Unary
@@ -2,14 +2,14 @@
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
#include "Interface/IR/IR.h"
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
#include <arm_neon.h>
#endif
#include <cstring>
namespace FEXCore::CPU {
#ifdef ARCHITECTURE_arm64
#ifdef _M_ARM_64
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
const auto is_using_words = (control & 1) != 0;
@@ -19,8 +19,8 @@ enum FallbackABI {
FABI_F80_I16_I32_PTR,
FABI_F32_I16_F80_PTR,
FABI_F64_I16_F80_PTR,
FABI_F64_F64_PTR,
FABI_F64_F64_F64_PTR,
FABI_F64_I16_F64_PTR,
FABI_F64_I16_F64_F64_PTR,
FABI_I16_I16_F80_PTR,
FABI_I32_I16_F80_PTR,
FABI_I64_I16_F80_PTR,
@@ -28,7 +28,7 @@ enum FallbackABI {
FABI_F80_I16_F80_PTR,
FABI_F80_I16_F80_F80_PTR,
FABI_F80x2_I16_F80_PTR,
FABI_F64x2_F64_PTR,
FABI_F64x2_I16_F64_PTR,
FABI_I32_I64_I64_V128_V128_I16,
FABI_I32_V128_V128_I16,
FABI_UNKNOWN,
+57 -109
View File
@@ -43,28 +43,21 @@ DEF_BINOP_WITH_CONSTANT(Ror, rorv, ror)
DEF_OP(Constant) {
auto Op = IROp->C<IR::IROp_Constant>();
auto Dst = GetReg(Node);
const auto PadType = [Pad = Op->Pad]() {
switch (Pad) {
case IR::ConstPad::NoPad: return CPU::Arm64Emitter::PadType::NOPAD;
case IR::ConstPad::DoPad: return CPU::Arm64Emitter::PadType::DOPAD;
default: return CPU::Arm64Emitter::PadType::AUTOPAD;
}
}();
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Op->Constant, PadType, Op->MaxBytes);
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Op->Constant);
}
DEF_OP(EntrypointOffset) {
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
auto Constant = Entry + Op->Offset;
auto Dst = GetReg(Node);
uint64_t Mask = ~0ULL;
const auto OpSize = IROp->Size;
if (OpSize == IR::OpSize::i32Bit) {
Mask = 0xFFFF'FFFFULL;
}
InsertGuestRIPMove(GetReg(Node), Constant & Mask);
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Constant & Mask);
}
DEF_OP(InlineConstant) {
@@ -379,7 +372,7 @@ DEF_OP(CondSubNZCV) {
DEF_OP(Neg) {
auto Op = IROp->C<IR::IROp_Neg>();
if (Op->Cond == IR::CondClass::AL) {
if (Op->Cond == FEXCore::IR::COND_AL) {
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src));
} else {
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src), MapCC(Op->Cond));
@@ -522,12 +515,6 @@ DEF_OP(AndWithFlags) {
}
}
DEF_OP(AndShift) {
auto Op = IROp->C<IR::IROp_XorShift>();
and_(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
DEF_OP(XorShift) {
auto Op = IROp->C<IR::IROp_XorShift>();
@@ -595,7 +582,7 @@ DEF_OP(ShiftFlags) {
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, TMP1, &Done);
cbz(EmitSize, TMP1, &Done);
{
// PF/SF/ZF/OF
if (OpSize >= IR::OpSize::i32Bit) {
@@ -659,7 +646,7 @@ DEF_OP(ShiftFlags) {
msr(ARMEmitter::SystemRegister::NZCV, TMP2);
}
}
(void)Bind(&Done);
Bind(&Done);
// TODO: Make RA less dumb so this can't happen (e.g. with late-kill).
if (PFOutput != PFTemp) {
@@ -676,7 +663,7 @@ DEF_OP(RotateFlags) {
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, Shift, &Done);
cbz(EmitSize, Shift, &Done);
{
// Extract the last bit shifted in to CF
const auto BitSize = IR::OpSizeToSize(Op->Size) * 8;
@@ -708,7 +695,7 @@ DEF_OP(RotateFlags) {
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
}
}
(void)Bind(&Done);
Bind(&Done);
}
DEF_OP(Extr) {
@@ -774,14 +761,14 @@ DEF_OP(PDep) {
// Now, they're copied, so we can start setting Dest (even if it overlaps with
// one of them). Handle early exit case
mov(EmitSize, Dest, 0);
(void)cbz(EmitSize, OrigMask, &Done);
cbz(EmitSize, OrigMask, &Done);
// Setup for first iteration
neg(EmitSize, T0, Mask);
and_(EmitSize, T0, T0, Mask);
// Main loop
(void)Bind(&NextBit);
Bind(&NextBit);
sbfx(EmitSize, T1, Input, 0, 1);
eor(EmitSize, Mask, Mask, T0);
and_(EmitSize, T0, T1, T0);
@@ -789,10 +776,10 @@ DEF_OP(PDep) {
orr(EmitSize, Dest, Dest, T0);
lsr(EmitSize, Input, Input, 1);
and_(EmitSize, T0, Mask, T1);
(void)cbnz(EmitSize, T0, &NextBit);
cbnz(EmitSize, T0, &NextBit);
// All done with nothing to do.
(void)Bind(&Done);
Bind(&Done);
}
}
@@ -828,27 +815,27 @@ DEF_OP(PExt) {
ARMEmitter::BackwardLabel NextBit;
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, Mask, &EarlyExit);
cbz(EmitSize, Mask, &EarlyExit);
mov(EmitSize, MaskReg, Mask);
mov(EmitSize, ValueReg, Input);
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
// Main loop
(void)Bind(&NextBit);
(void)cbz(EmitSize, MaskReg, &Done);
Bind(&NextBit);
cbz(EmitSize, MaskReg, &Done);
clz(EmitSize, BitReg, MaskReg);
lslv(EmitSize, ValueReg, ValueReg, BitReg);
lslv(EmitSize, MaskReg, MaskReg, BitReg);
extr(EmitSize, Dest, Dest, ValueReg, OpSizeBitsM1);
bfc(EmitSize, MaskReg, OpSizeBitsM1, 1);
(void)b(&NextBit);
b(&NextBit);
// Early exit
(void)Bind(&EarlyExit);
Bind(&EarlyExit);
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
// All done with nothing to do.
(void)Bind(&Done);
Bind(&Done);
}
}
@@ -916,7 +903,7 @@ DEF_OP(Div) {
eor(EmitSize, TMP1, TMP1, Upper);
// If the sign bit matches then the result is zero
(void)cbz(EmitSize, TMP1, &Only64Bit);
cbz(EmitSize, TMP1, &Only64Bit);
// Long divide
{
@@ -924,7 +911,7 @@ DEF_OP(Div) {
mov(EmitSize, TMP2, Lower);
mov(EmitSize, TMP3, Divisor);
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.LDIVHandler));
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP4);
@@ -935,17 +922,17 @@ DEF_OP(Div) {
mov(EmitSize, Remainder, TMP2);
// Skip 64-bit path
(void)b(&LongDIVRet);
b(&LongDIVRet);
}
(void)Bind(&Only64Bit);
Bind(&Only64Bit);
// 64-Bit only
{
sdiv(EmitSize, Quotient, Lower, Divisor);
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
}
(void)Bind(&LongDIVRet);
Bind(&LongDIVRet);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown DIV Size: {}", OpSize); break;
@@ -999,7 +986,7 @@ DEF_OP(UDiv) {
// Check the upper bits for zero
// If the upper bits are zero then we can do a 64-bit divide
(void)cbz(EmitSize, Upper, &Only64Bit);
cbz(EmitSize, Upper, &Only64Bit);
// Long divide
{
@@ -1007,7 +994,7 @@ DEF_OP(UDiv) {
mov(EmitSize, TMP2, Lower);
mov(EmitSize, TMP3, Divisor);
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.LUDIVHandler));
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
blr(TMP4);
@@ -1018,17 +1005,17 @@ DEF_OP(UDiv) {
mov(EmitSize, Remainder, TMP2);
// Skip 64-bit path
(void)b(&LongDIVRet);
b(&LongDIVRet);
}
(void)Bind(&Only64Bit);
Bind(&Only64Bit);
// 64-Bit only
{
udiv(EmitSize, Quotient, Lower, Divisor);
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
}
(void)Bind(&LongDIVRet);
Bind(&LongDIVRet);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUDIV Size: {}", OpSize); break;
@@ -1051,50 +1038,34 @@ DEF_OP(Popcount) {
const auto Dst = GetReg(Node);
const auto Src = GetReg(Op->Src);
if (CTX->HostFeatures.SupportsCSSC) {
switch (OpSize) {
case IR::OpSize::i8Bit:
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i16Bit:
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i32Bit: cnt(ARMEmitter::Size::i32Bit, Dst, Src); break;
case IR::OpSize::i64Bit: cnt(ARMEmitter::Size::i64Bit, Dst, Src); break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
}
} else {
switch (OpSize) {
case IR::OpSize::i8Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
// only use lowest byte
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i16Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// only count two lowest bytes
addp(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i32Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// fmov has zero extended, unused bytes are zero
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i64Bit:
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// fmov has zero extended, unused bytes are zero
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
}
umov<ARMEmitter::SubRegSize::i8Bit>(Dst, VTMP1, 0);
switch (OpSize) {
case IR::OpSize::i8Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
// only use lowest byte
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i16Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// only count two lowest bytes
addp(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i32Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// fmov has zero extended, unused bytes are zero
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i64Bit:
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// fmov has zero extended, unused bytes are zero
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
}
umov<ARMEmitter::SubRegSize::i8Bit>(Dst, VTMP1, 0);
}
DEF_OP(FindLSB) {
@@ -1197,19 +1168,6 @@ DEF_OP(Rev) {
}
}
DEF_OP(Rbit) {
auto Op = IROp->C<IR::IROp_Rbit>();
const auto OpSize = IROp->Size;
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
const auto EmitSize = ConvertSize48(IROp);
const auto Dst = GetReg(Node);
const auto Src = GetReg(Op->Src);
rbit(EmitSize, Dst, Src);
}
DEF_OP(Bfi) {
auto Op = IROp->C<IR::IROp_Bfi>();
const auto EmitSize = ConvertSize(IROp);
@@ -1293,16 +1251,6 @@ DEF_OP(Sbfe) {
sbfx(ConvertSize(IROp), Dst, Src, Op->lsb, Op->Width);
}
DEF_OP(MaskGenerateFromBitWidth) {
auto Op = IROp->C<IR::IROp_MaskGenerateFromBitWidth>();
auto BitWidth = GetReg(Op->BitWidth);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1);
cmp(ARMEmitter::Size::i64Bit, BitWidth, 0);
lslv(ARMEmitter::Size::i64Bit, TMP2, TMP1, BitWidth);
csinv(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1, TMP2, ARMEmitter::Condition::CC_EQ);
}
DEF_OP(Select) {
auto Op = IROp->C<IR::IROp_Select>();
const auto OpSize = IROp->Size;
@@ -1399,12 +1347,12 @@ DEF_OP(VExtractToGPR) {
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
const auto OpSize = IROp->Size;
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
[[maybe_unused]] constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
const auto ElementSizeBits = IR::OpSizeAsBits(Op->Header.ElementSize);
const auto Offset = ElementSizeBits * Op->Index;
const auto Is256Bit = Offset >= SSERegBitSize;
[[maybe_unused]] const auto Is256Bit = Offset >= SSERegBitSize;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetReg(Node);
@@ -11,33 +11,36 @@ $end_info$
#include <FEXCore/Core/Thunks.h>
namespace FEXCore::CPU {
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl& CTX, FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
switch (Op) {
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return CTX.Dispatcher->GetExitFunctionLinkerAddress();
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
break;
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op)); break;
}
return ~0ULL;
}
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum) {
Relocation MoveABI {};
MoveABI.NamedThunkMove.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE};
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
MoveABI.NamedThunkMove.Offset = CurrentCursor - CodeData.BlockBegin;
MoveABI.NamedThunkMove.Symbol = Sum;
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
// Pointers are required to fit within 48-bit VA space.
// TODO: Force 6-byte `MaxSize`, with zext extension to 64-bit. Current code not smart enough to handle negatives.
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, FEXCore::CPU::Arm64Emitter::PadType::AUTOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
uint64_t Pointer = GetNamedSymbolLiteral(*CTX, Op);
uint64_t Pointer = GetNamedSymbolLiteral(Op);
NamedSymbolLiteralPair Lit {
Arm64JITCore::NamedSymbolLiteralPair Lit {
.Lit = Pointer,
.MoveABI =
{
@@ -45,77 +48,87 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
{
.Header =
{
.Offset = 0, // Set by PlaceNamedSymbolLiteral
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
},
.Symbol = Op,
.Offset = 0,
},
},
};
return Lit;
}
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit) {
switch (Lit.MoveABI.Header.Type) {
case RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL:
case RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
Lit.MoveABI.Header.Offset = GetCursorOffset();
break;
}
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
default: ERROR_AND_DIE_FMT("Unknown relocation type for {}", __FUNCTION__);
}
BindOrRestart(&Lit.Loc);
Bind(&Lit.Loc);
dc64(Lit.Lit);
Relocations.emplace_back(Lit.MoveABI);
}
auto Arm64JITCore::InsertGuestRIPLiteral(uint64_t GuestRIP) -> NamedSymbolLiteralPair {
return {
.Lit = GuestRIP,
.MoveABI =
{
.GuestRIP = {.Header =
{
.Offset = 0, // Set by PlaceNamedSymbolLiteral
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL,
},
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
.GuestRIP = GuestRIP},
},
};
}
void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant) {
Relocation MoveABI {};
MoveABI.GuestRIP.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE};
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
MoveABI.GuestRIP.GuestRIP = Constant;
MoveABI.GuestRIP.RegisterIndex = Reg.Idx();
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
MoveABI.GuestRIPMove.Offset = CurrentCursor - CodeData.BlockBegin;
MoveABI.GuestRIPMove.GuestRIP = Constant;
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
// Pointers are required to fit within 48-bit VA space.
// TODO: Force 6-byte `MaxSize`, with sign extension to 64-bit. Current code not smart enough to handle negatives.
// 48-bit sign extension works because x86-64 guests only receive 47-bit VA space, with 48-bit being reserved for kernel.
// Additional quirk, "canonical" 48-bit pointers on x86-64, sign extend the 48-bit as well (Which is why kernel pointers are negative).
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, FEXCore::CPU::Arm64Emitter::PadType::AUTOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations(uint64_t GuestBaseAddress) {
// Rebase relocations to library base address
for (auto& Relocation : Relocations) {
switch (Relocation.Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE:
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
Relocation.GuestRIP.GuestRIP -= GuestBaseAddress;
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
const char* EntryRelocations) {
size_t DataIndex {};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
dc64(Pointer);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
// XXX: Reenable once the JIT Object Cache is upstream
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
default:;
}
}
return std::move(Relocations);
return true;
}
} // namespace FEXCore::CPU
Loaded 100 of 1241 files, more files were not shown because too many files have changed in this diff. Show more