mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 18:00:17 +02:00
Compare commits
17
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a141d8bd93 | ||
|
|
52e21a6e02 | ||
|
|
7eb4520317 | ||
|
|
d4515c3a6c | ||
|
|
d214ebc8f2 | ||
|
|
c379eede3b | ||
|
|
8ea92ab9b6 | ||
|
|
40d9c66784 | ||
|
|
cc4da669c9 | ||
|
|
22a58925c7 | ||
|
|
25ed2578c2 | ||
|
|
90dcfab131 | ||
|
|
91ac4c9a3a | ||
|
|
0d86ee575b | ||
|
|
0409698783 | ||
|
|
ec40d53cc9 | ||
|
|
d46e6fac22 |
No files matched your search
@@ -2,11 +2,8 @@
|
||||
Source/Common/cpp-optparse/*
|
||||
|
||||
# Files with human-indented tables for readability - don't mess with these
|
||||
FEXCore/Source/Interface/Core/X86Tables/*.cpp
|
||||
FEXCore/Source/Interface/Core/X86Tables/*
|
||||
|
||||
# Inline headers with list-like content that can't be processed individually
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
|
||||
|
||||
# Include files in unittests
|
||||
unittests/*ASM/Includes/*.inc
|
||||
@@ -22,6 +22,3 @@
|
||||
|
||||
# Minor reformat with clang-format-19
|
||||
9fdd96af61c969cb5732471223f00eda64b7a069
|
||||
|
||||
# Reformat of X86Tables.h
|
||||
ba2b0ef809f66f1a6d334f000798fa2ceafab26f
|
||||
+195
-96
@@ -24,139 +24,238 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: '0'
|
||||
fetch-tags: 'true'
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner info
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Setup Build Environment
|
||||
uses: ./.github/workflows/setup-env
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
run: |
|
||||
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True \
|
||||
-DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True \
|
||||
-DCMAKE_INSTALL_PREFIX="$PWD"/build/install
|
||||
|
||||
# These steps make a lot of noise but rarely fail.
|
||||
# Put them in a separate step to make normal build logs easier to parse
|
||||
- name: Noisy Build Targets
|
||||
run: cmake --build build --target asm_files 32bit_asm_files JemallocLibs Catch2 vixl cephes_128bit
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
id: build
|
||||
run: cmake --build build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: Install
|
||||
run: cmake --build build --target install
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
# GCC tests
|
||||
- name: GCC64 Target Tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: gcc_target_tests_64
|
||||
- name: gcc target tests 64
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_64
|
||||
|
||||
- name: GCC32 Target Tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: gcc_target_tests_32
|
||||
- name: GCC64 Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
|
||||
|
||||
# API tests
|
||||
- name: API Tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: api_tests
|
||||
- name: gcc target tests 32
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
|
||||
|
||||
- name: FEXCore API Tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: fexcore_apitests
|
||||
- name: GCC32 Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
|
||||
|
||||
# ARM emission tests
|
||||
- name: ARM Emitter Tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: emitter_tests
|
||||
- name: APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target api_tests
|
||||
|
||||
# Linux tests
|
||||
- name: FEX Linux Tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: fex_linux_tests_all
|
||||
- name: APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXCore APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
|
||||
|
||||
- name: FEXCore APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
|
||||
|
||||
- name: ARMEmitter tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target emitter_tests
|
||||
|
||||
- name: ARMEmitter Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ARMEmitterTests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
# These tests require non-portable install due to thunks.
|
||||
FEX_PORTABLE: 0
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
|
||||
|
||||
# Thunking
|
||||
- name: Thunkgen tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: thunkgen_tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunkgen_tests
|
||||
|
||||
- name: Thunkgen Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
|
||||
|
||||
- name: Test GL No-Thunks
|
||||
if: ${{ steps.build.outcome == 'success' && matrix.arch[1] == 'x64' }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: thunk_functional_tests_nothunks
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DISPLAY: ':0'
|
||||
DISPLAY: ":0"
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_nothunks
|
||||
|
||||
- name: No thunks Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_NoThunkResults.log || true
|
||||
|
||||
- name: Test GL Thunks
|
||||
if: ${{ steps.build.outcome == 'success' && matrix.arch[1] == 'x64' }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: thunk_functional_tests_thunks
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DISPLAY: ':0'
|
||||
DISPLAY: ":0"
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_thunks
|
||||
|
||||
- name: Thunks Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkResults.log || true
|
||||
|
||||
# ASM tests
|
||||
- name: ASM Tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: asm_tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
# POSIX tests
|
||||
- name: POSIX Tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: posix_tests
|
||||
- name: ASM Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
|
||||
# GVisor tests
|
||||
- name: GVisor Tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: gvisor_tests
|
||||
- name: Posix Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the posixtest
|
||||
run: cmake --build . --config $BUILD_TYPE --target posix_tests
|
||||
|
||||
- name: Posix Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
|
||||
|
||||
- name: gvisor tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gvisor_tests
|
||||
|
||||
- name: GVisor Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GVisor.log || true
|
||||
|
||||
# Struct verifier tests
|
||||
- name: Struct verifier tests
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: struct_verifier
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target struct_verifier
|
||||
|
||||
- name: Struct verifier Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
run: cmake --build build --target remove_old_shm_regions
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
|
||||
path: results/*.log
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
@@ -31,94 +31,163 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: '0'
|
||||
fetch-tags: 'true'
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner info
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Setup Build Environment
|
||||
uses: ./.github/workflows/setup-env
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
run: |
|
||||
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False \
|
||||
-DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True \
|
||||
-DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False \
|
||||
-DCMAKE_INSTALL_PREFIX="$PWD"/build/install
|
||||
|
||||
# These steps make a lot of noise but rarely fail.
|
||||
# Put them in a separate step to make normal build logs easier to parse
|
||||
- name: Noisy Build Targets
|
||||
run: cmake --build build --target asm_files 32bit_asm_files JemallocLibs Catch2 vixl cephes_128bit
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: Install
|
||||
run: cmake --build build --target install
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
# GCC tests
|
||||
- name: GCC64 Target Tests
|
||||
- name: gcc target tests 64
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_64
|
||||
|
||||
- name: GCC64 Test Results move
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: gcc_target_tests_64
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
|
||||
|
||||
- name: GCC32 Target Tests
|
||||
- name: gcc target tests 32
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
|
||||
|
||||
- name: GCC32 Test Results move
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: gcc_target_tests_32
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
|
||||
|
||||
# API Tests
|
||||
- name: API Tests
|
||||
- name: APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target api_tests
|
||||
|
||||
- name: APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: api_tests
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXCore API Tests
|
||||
- name: FEXCore APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
|
||||
|
||||
- name: FEXCore APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: fexcore_apitests
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
|
||||
|
||||
# Linux tests
|
||||
- name: FEX Linux Tests
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: fex_linux_tests_all
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
|
||||
|
||||
# ASM Tests
|
||||
- name: ASM Tests
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: asm_tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
# POSIX Tests
|
||||
- name: POSIX Tests
|
||||
- name: ASM Test Results move
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: posix_tests
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
|
||||
- name: Posix Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the posixtest
|
||||
run: cmake --build . --config $BUILD_TYPE --target posix_tests
|
||||
|
||||
- name: Posix Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
run: cmake --build build --target remove_old_shm_regions
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
|
||||
path: results/*.log
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -24,45 +24,83 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: '0'
|
||||
fetch-tags: 'true'
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner info
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Setup Build Environment
|
||||
uses: ./.github/workflows/setup-env
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
run: |
|
||||
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False \
|
||||
-DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
# These steps make a lot of noise but rarely fail.
|
||||
# Put them in a separate step to make normal build logs easier to parse
|
||||
- name: Noisy Build Targets
|
||||
run: cmake --build build --target asm_files 32bit_asm_files JemallocLibs Catch2 vixl cephes_128bit
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
# ASM tests
|
||||
- name: ASM Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test Results move
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: asm_tests
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
|
||||
path: results/*.log
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -23,56 +23,123 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: '0'
|
||||
fetch-tags: 'true'
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner info
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Setup Build Environment
|
||||
uses: ./.github/workflows/setup-env
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name: Set VIXL_SIM_ENABLED
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
case '${{ matrix.arch[1] }}' in
|
||||
x64) _sim=True ;;
|
||||
ARM64) _sim=False ;;
|
||||
esac
|
||||
echo "VIXL_SIM_ENABLED=$_sim" >> $GITHUB_ENV
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Set vixl_sim x86
|
||||
if: matrix.arch[1] == 'x64'
|
||||
run: |
|
||||
echo "VIXL_SIM_ENABLED=True" >> $GITHUB_ENV
|
||||
|
||||
- name: Set vixl_sim Arm64
|
||||
if: matrix.arch[1] == 'ARM64'
|
||||
run: |
|
||||
echo "VIXL_SIM_ENABLED=False" >> $GITHUB_ENV
|
||||
|
||||
- name: Configure CMake
|
||||
run: |
|
||||
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED \
|
||||
-DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
FEX_DISABLETELEMETRY: 1
|
||||
run: cmake --build build --target CodeSizeValidation instcountci_test_files
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE --target CodeSizeValidation instcountci_test_files
|
||||
|
||||
- name: Instruction Count Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target instcountci_tests
|
||||
|
||||
- name: Instruction Count Test Results move
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: instcountci_tests
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_InstCountCI.log || true
|
||||
|
||||
- name: Update local repo instcount
|
||||
if: ${{ always() }}
|
||||
run: cmake --build build --target instcountci_update_tests
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target instcountci_update_tests
|
||||
|
||||
- name: Check InstCountCI diff
|
||||
- name: Get instcountCI diff
|
||||
if: ${{ always() }}
|
||||
run: git --no-pager diff --exit-code HEAD
|
||||
shell: bash
|
||||
working-directory: ${{github.workspace}}/
|
||||
run: git diff --output=${{runner.workspace}}/build/InstCountCI.diff
|
||||
|
||||
- name: Check if InstCountCI Diff exists
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{github.workspace}}/
|
||||
# Check if the file is empty
|
||||
run: sh -c "! test -s ${{runner.workspace}}/build/InstCountCI.diff"
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
|
||||
path: results/*.log
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
- name: Upload results InstCountCI
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}-instcountci
|
||||
path: ${{runner.workspace}}/build/InstCountCI.diff
|
||||
retention-days: 3
|
||||
|
||||
@@ -20,10 +20,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: '0'
|
||||
fetch-tags: 'true'
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -31,23 +28,73 @@ jobs:
|
||||
- name: Add MingGW to PATH
|
||||
run: echo "$HOME/llvm-mingw/build/bin/" >> $GITHUB_PATH
|
||||
|
||||
- name: Set CC
|
||||
- name: Set CC x86
|
||||
if: matrix.arch[1] == 'x64'
|
||||
run: |
|
||||
case '${{ matrix.arch[1] }}' in
|
||||
x64) _cpu=x86_64 ;;
|
||||
ARM64) _cpu=aarch64 ;;
|
||||
ARM64EC) _cpu=arm64ec ;;
|
||||
esac
|
||||
echo "MINGW_TRIPLE=${_cpu}-w64-mingw32" >> $GITHUB_ENV
|
||||
echo "MINGW_TRIPLE=x86_64-w64-mingw32" >> $GITHUB_ENV
|
||||
|
||||
- name: Setup Build Environment
|
||||
uses: ./.github/workflows/setup-env
|
||||
- name: Set CC Arm64
|
||||
if: matrix.arch[1] == 'ARM64'
|
||||
run: |
|
||||
echo "MINGW_TRIPLE=aarch64-w64-mingw32" >> $GITHUB_ENV
|
||||
|
||||
- name: Set CC Arm64EC
|
||||
if: matrix.arch[1] == 'ARM64EC'
|
||||
run: |
|
||||
echo "MINGW_TRIPLE=arm64ec-w64-mingw32" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
run: |
|
||||
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake \
|
||||
-DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False \
|
||||
-DCMAKE_INSTALL_PREFIX="$PWD"/build/install
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# Inspired by LLVM's pr-code-format.yml at
|
||||
# Inspired by LLVM's pr-code-format.yml at
|
||||
# https://github.com/llvm/llvm-project/blob/main/.github/workflows/pr-code-format.yml
|
||||
|
||||
name: Check code formatting
|
||||
name: "Check code formatting"
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
@@ -13,7 +13,7 @@ jobs:
|
||||
if: github.repository == 'FEX-Emu/FEX'
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
- name: Fetch FEX sources
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ github.event.pull_request.head.sha }}
|
||||
@@ -27,13 +27,18 @@ jobs:
|
||||
deepen_length: 500
|
||||
|
||||
- name: Get changed files
|
||||
run: |
|
||||
BASE=$(git merge-base main HEAD)
|
||||
FILES=$(git diff --name-only "$BASE" | tr '\n' ',' | sed 's/,$//')
|
||||
echo "CHANGED_FILES=$FILES" >> $GITHUB_ENV
|
||||
id: changed-files
|
||||
uses: step-security/changed-files@3dbe17c78367e7d60f00d78ae6781a35be47b4a1 # v45.0.1
|
||||
with:
|
||||
separator: ","
|
||||
skip_initial_fetch: true
|
||||
|
||||
echo "Changed files:"
|
||||
echo "$FILES"
|
||||
- name: "Listed files"
|
||||
env:
|
||||
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
|
||||
run: |
|
||||
echo "Formatting files:"
|
||||
echo "$CHANGED_FILES"
|
||||
|
||||
- name: Check git-clang-format-19 exists
|
||||
run: which git-clang-format-19
|
||||
@@ -41,23 +46,24 @@ jobs:
|
||||
- name: Setup Python env
|
||||
uses: actions/setup-python@v4
|
||||
with:
|
||||
python-version: 3.11
|
||||
cache: pip
|
||||
cache-dependency-path: ./External/code-format-helper/requirements_formatting.txt
|
||||
python-version: '3.11'
|
||||
cache: 'pip'
|
||||
cache-dependency-path: './External/code-format-helper/requirements_formatting.txt'
|
||||
|
||||
- name: Install python dependencies
|
||||
run: pip install -r ./External/code-format-helper/requirements_formatting.txt
|
||||
|
||||
- name: Run code formatter
|
||||
env:
|
||||
CLANG_FORMAT_PATH: git-clang-format-19
|
||||
CLANG_FORMAT_PATH: 'git-clang-format-19'
|
||||
GITHUB_PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
START_REV: ${{ github.event.pull_request.base.sha }}
|
||||
END_REV: ${{ github.event.pull_request.head.sha }}
|
||||
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
|
||||
run: |
|
||||
python ./External/code-format-helper/code-format-helper.py \
|
||||
--repo "FEX-Emu/FEX" \
|
||||
--issue-number "$GITHUB_PR_NUMBER" \
|
||||
--start-rev "$START_REV" \
|
||||
--end-rev "$END_REV" \
|
||||
--repo "FEX-emu/FEX" \
|
||||
--issue-number $GITHUB_PR_NUMBER \
|
||||
--start-rev $START_REV \
|
||||
--end-rev $END_REV \
|
||||
--changed-files "$CHANGED_FILES"
|
||||
@@ -1,33 +0,0 @@
|
||||
name: Setup Build Environment
|
||||
description: Setup RootFS and build environment
|
||||
|
||||
inputs:
|
||||
setup-rootfs:
|
||||
description: 'Whether or not to set up the rootfs'
|
||||
default: true
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Set rootfs paths
|
||||
if: ${{ inputs.setup-rootfs == 'true' }}
|
||||
shell: bash
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
if: ${{ inputs.setup-rootfs == 'true' }}
|
||||
shell: bash
|
||||
run: python3 Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name: Checkout Submodules
|
||||
shell: bash
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
shell: bash
|
||||
run: rm -Rf build
|
||||
@@ -1,72 +0,0 @@
|
||||
name: steamrt4 build
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
DEBIAN_FRONTEND: noninteractive
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
|
||||
jobs:
|
||||
steamrt4_build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
arch: [[self-hosted, ARM64, distrobox]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: '0'
|
||||
fetch-tags: 'true'
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Setup Build Environment
|
||||
uses: ./.github/workflows/setup-env
|
||||
with:
|
||||
setup-rootfs: false
|
||||
|
||||
# Setup everything required.
|
||||
- name : distrobox setup
|
||||
run: |
|
||||
distrobox create -Y -i registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306 steamrt4 || true
|
||||
distrobox upgrade steamrt4
|
||||
distrobox enter --name steamrt4 -- sudo apt-get install -y \
|
||||
git cmake ninja-build ccache \
|
||||
lld clang clang-tools \
|
||||
libclang-dev llvm-dev \
|
||||
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross \
|
||||
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
|
||||
|
||||
- name: Configure CMake
|
||||
run: |
|
||||
distrobox enter --name steamrt4 -- cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE \
|
||||
-G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True \
|
||||
-DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld \
|
||||
-DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Build
|
||||
run: distrobox enter --name steamrt4 -- cmake --build build
|
||||
|
||||
- name: install
|
||||
run: DESTDIR="$PWD"/install distrobox enter --name steamrt4 -- cmake --build build -t install
|
||||
|
||||
- name: Upload libraries
|
||||
uses: actions/upload-artifact@v6
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
overwrite: true
|
||||
name: steamrt4_steampipe_depot
|
||||
path: ${{ github.workspace }}/install/*
|
||||
retention-days: 60
|
||||
compression-level: 9
|
||||
@@ -1,21 +0,0 @@
|
||||
name: Run Test and Store Logs
|
||||
description: Run a test and store the log.
|
||||
inputs:
|
||||
target:
|
||||
description: 'The test target to run'
|
||||
required: true
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Run Tests
|
||||
shell: bash
|
||||
run: cmake --build build --target ${{ inputs.target }}
|
||||
|
||||
- name: Move and Truncate Results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
run: |
|
||||
mkdir -p results
|
||||
mv build/Testing/Temporary/LastTest.log results/${{ inputs.target }}.log || true
|
||||
truncate --size="<20M" results/${{ inputs.target }}.log || true
|
||||
@@ -25,59 +25,111 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: '0'
|
||||
fetch-tags: 'true'
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner info
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Setup Build Environment
|
||||
uses: ./.github/workflows/setup-env
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
run: |
|
||||
cmake -S . -B build -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_LTO=False \
|
||||
-DENABLE_VIXL_DISASSEMBLER=True -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
# These steps make a lot of noise but rarely fail.
|
||||
# Put them in a separate step to make normal build logs easier to parse
|
||||
- name: Noisy Build Targets
|
||||
run: cmake --build build --target asm_files 32bit_asm_files JemallocLibs Catch2 vixl cephes_128bit
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: ASM Tests - SVE256
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test SVE256 Results move
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: asm_tests
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM_SVE256Bit.log || true
|
||||
|
||||
- name: ASM Tests - SVE128
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
FEX_FORCESVEWIDTH: "128"
|
||||
with:
|
||||
target: asm_tests
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test 128-bit Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM_SVE128Bit.log || true
|
||||
|
||||
- name: ASM Tests - ASIMD
|
||||
if: ${{ always() }}
|
||||
uses: ./.github/workflows/test
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
FEX_HOSTFEATURES: "disablesve"
|
||||
with:
|
||||
target: asm_tests
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test ASIMD Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM_ASIMD.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}-${{ env.runner_label }}
|
||||
path: results/*.log
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -1,49 +0,0 @@
|
||||
name: Wine DLL Build
|
||||
description: Build a wow64 or arm64ec Wine DLL
|
||||
|
||||
inputs:
|
||||
target:
|
||||
description: 'The target (arm64ec or wow64)'
|
||||
required: true
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Clean Build Environment
|
||||
shell: bash
|
||||
run: rm -Rf build_${{ inputs.target }}
|
||||
|
||||
- name: Configure CMake
|
||||
shell: bash
|
||||
run: |
|
||||
case "${{ inputs.target }}" in
|
||||
wow64) _cc=aarch64 ;;
|
||||
arm64ec) _cc=arm64ec ;;
|
||||
esac
|
||||
|
||||
cmake -S . -B build_${{ inputs.target }} -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=Data/CMake/toolchain_mingw.cmake \
|
||||
-DMINGW_TRIPLE=${_cc}-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja \
|
||||
-DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False \
|
||||
-DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=generic -DTUNE_CPU=none -DRANGES_NATIVE=OFF
|
||||
|
||||
- name: Build
|
||||
shell: bash
|
||||
run: cmake --build build_${{ inputs.target }}
|
||||
|
||||
- name: Install
|
||||
shell: bash
|
||||
run: DESTDIR="$PWD"/install cmake --build build_${{ inputs.target }} -t install
|
||||
|
||||
- name: Configure UnixLib
|
||||
shell: bash
|
||||
run: |
|
||||
cmake -S Source/Windows/UnixLib -B build_unixlib_${{ inputs.target }} -DCMAKE_BUILD_TYPE=$BUILD_TYPE \
|
||||
-G Ninja -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-unix -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Build UnixLib
|
||||
shell: bash
|
||||
run: cmake --build build_unixlib_${{ inputs.target }}
|
||||
|
||||
- name: Install UnixLib
|
||||
shell: bash
|
||||
run: DESTDIR="$PWD"/install cmake --build build_unixlib_${{ inputs.target }} -t install
|
||||
@@ -17,41 +17,72 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: '0'
|
||||
fetch-tags: 'true'
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Add MingGW to PATH
|
||||
run: echo "$HOME/llvm-mingw/build/bin/" >> $GITHUB_PATH
|
||||
|
||||
- name: Checkout Submodules
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean install directory
|
||||
run: rm -Rf install
|
||||
run: |
|
||||
rm -Rf ${{runner.workspace}}/build_install
|
||||
mkdir ${{runner.workspace}}/build_install
|
||||
|
||||
- name: Build (wow64)
|
||||
uses: ./.github/workflows/wine_build
|
||||
with:
|
||||
target: wow64
|
||||
- name: Clean Build Environment
|
||||
run: |
|
||||
rm -Rf ${{runner.workspace}}/build_arm64ec
|
||||
rm -Rf ${{runner.workspace}}/build_wow64
|
||||
|
||||
- name: Build (arm64ec)
|
||||
uses: ./.github/workflows/wine_build
|
||||
with:
|
||||
target: arm64ec
|
||||
- name: Create Build Environment arm64ec
|
||||
run: |
|
||||
cmake -E make_directory ${{runner.workspace}}/build_arm64ec
|
||||
cmake -E make_directory ${{runner.workspace}}/build_wow64
|
||||
|
||||
- name: Configure CMake arm64ec
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build_arm64ec
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Configure CMake wow64
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build_wow64
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Build arm64ec
|
||||
working-directory: ${{runner.workspace}}/build_arm64ec
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: install arm64ec
|
||||
working-directory: ${{runner.workspace}}/build_arm64ec
|
||||
shell: bash
|
||||
env:
|
||||
DESTDIR: ${{runner.workspace}}/build_install
|
||||
run: cmake --build . --config $BUILD_TYPE -t install
|
||||
|
||||
- name: Build wow64
|
||||
working-directory: ${{runner.workspace}}/build_wow64
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: install wow64
|
||||
working-directory: ${{runner.workspace}}/build_wow64
|
||||
shell: bash
|
||||
env:
|
||||
DESTDIR: ${{runner.workspace}}/build_install
|
||||
run: cmake --build . --config $BUILD_TYPE -t install
|
||||
|
||||
- name: Upload libraries
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
overwrite: true
|
||||
name: wine_dll_artifacts
|
||||
path: |
|
||||
${{ github.workspace }}/install/usr/lib/wine/aarch64-windows/lib*.dll
|
||||
${{ github.workspace }}/install/usr/lib/wine/aarch64-unix/lib*.so
|
||||
path: ${{runner.workspace}}/build_install/usr/lib/wine/aarch64-windows/lib*.dll
|
||||
retention-days: 60
|
||||
compression-level: 9
|
||||
@@ -11,5 +11,3 @@ out/
|
||||
.vs/
|
||||
*.pyc
|
||||
.cache
|
||||
.idea/
|
||||
CMakeLists.txt.user
|
||||
@@ -1,71 +0,0 @@
|
||||
spec:
|
||||
inputs:
|
||||
PROMOTE_BRANCH:
|
||||
description: "Branch to promote the build to. Empty means no promotion."
|
||||
default: "bleeding-edge"
|
||||
|
||||
---
|
||||
|
||||
workflow:
|
||||
rules:
|
||||
- when: always
|
||||
variables:
|
||||
PROMOTE_BRANCH: $[[ inputs.PROMOTE_BRANCH ]]
|
||||
|
||||
variables:
|
||||
DEBIAN_FRONTEND: noninteractive
|
||||
GIT_SUBMODULE_STRATEGY: recursive
|
||||
GIT_DEPTH: 0
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
|
||||
build:
|
||||
stage: build
|
||||
image: registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306
|
||||
tags:
|
||||
- docker
|
||||
- linux
|
||||
- arm64
|
||||
- aarch64
|
||||
script:
|
||||
- apt-get -y update
|
||||
- apt-get install -y
|
||||
git cmake ninja-build ccache
|
||||
lld clang clang-tools
|
||||
libclang-dev llvm-dev
|
||||
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross
|
||||
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
|
||||
- cmake -E make_directory build/
|
||||
- cmake -DCMAKE_BUILD_TYPE=Release -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=armv8.2-a -DTUNE_CPU=none -DRANGES_NATIVE=OFF . -B build/
|
||||
- cmake --build build/ --config Release
|
||||
- DESTDIR=$(pwd)/install/ cmake --build build/ --config Release -t install
|
||||
|
||||
artifacts:
|
||||
name: "steamrt artifacts"
|
||||
untracked: false
|
||||
paths:
|
||||
- install/
|
||||
|
||||
promote:
|
||||
stage: deploy
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
image: registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306
|
||||
tags:
|
||||
- docker
|
||||
- linux
|
||||
- arm64
|
||||
- aarch64
|
||||
rules:
|
||||
- if: '$PROMOTE_BRANCH'
|
||||
before_script:
|
||||
- apt-get -y update
|
||||
- apt-get install -y tmux curl
|
||||
script:
|
||||
# comment out to debug: SSH in via GCP, go down the container and attach to the session (with `tmux attach -t debug`)
|
||||
# - tmux new-session -d -s debug
|
||||
# - while tmux has-session -t debug 2>/dev/null; do sleep 1; done
|
||||
|
||||
# ref controls which fex-depot code runs the pipeline, while VERSION_PARAM controls which fex branch's artifacts that pipeline downloads.
|
||||
- >
|
||||
curl --fail --location --request POST --form token=${FEX_DEPOT_TRIGGER_TOKEN} --form ref=master --form "variables[PROMOTE_BRANCH]=${PROMOTE_BRANCH}" --form "variables[VERSION_PARAM]=${CI_COMMIT_REF_NAME}" "${CI_API_V4_URL}/projects/fex%2Ffex-depot/trigger/pipeline"
|
||||
+7
-10
@@ -17,6 +17,9 @@
|
||||
shallow = true
|
||||
path = External/fex-gcc-target-tests-bins
|
||||
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
|
||||
[submodule "External/jemalloc"]
|
||||
path = External/jemalloc
|
||||
url = https://github.com/FEX-Emu/jemalloc.git
|
||||
[submodule "External/fmt"]
|
||||
path = External/fmt
|
||||
url = https://github.com/fmtlib/fmt.git
|
||||
@@ -29,6 +32,10 @@
|
||||
[submodule "External/Catch2"]
|
||||
path = External/Catch2
|
||||
url = https://github.com/catchorg/Catch2.git
|
||||
[submodule "External/robin-map"]
|
||||
shallow = true
|
||||
path = External/robin-map
|
||||
url = https://github.com/FEX-Emu/robin-map.git
|
||||
[submodule "External/Vulkan-Headers"]
|
||||
shallow = true
|
||||
path = External/Vulkan-Headers
|
||||
@@ -42,13 +49,3 @@
|
||||
[submodule "External/range-v3"]
|
||||
path = External/range-v3
|
||||
url = https://github.com/ericniebler/range-v3.git
|
||||
[submodule "External/zydis"]
|
||||
shallow = true
|
||||
path = External/zydis
|
||||
url = https://github.com/zyantific/zydis.git
|
||||
[submodule "External/unordered_dense"]
|
||||
path = External/unordered_dense
|
||||
url = https://github.com/martinus/unordered_dense.git
|
||||
[submodule "External/rpmalloc"]
|
||||
path = External/rpmalloc
|
||||
url = https://github.com/FEX-Emu/rpmalloc.git
|
||||
@@ -1 +0,0 @@
|
||||
AI must not be used to generate code for contributions to this project.
|
||||
@@ -1 +0,0 @@
|
||||
AI must not be used to generate code for contributions to this project.
|
||||
+180
-294
@@ -1,50 +1,46 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
project(FEX C CXX ASM)
|
||||
|
||||
include(CheckIncludeFiles)
|
||||
check_include_files("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
INCLUDE (CheckIncludeFiles)
|
||||
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests (requires x86 compiler)" FALSE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" TRUE)
|
||||
option(ENABLE_IWYU "Enable the Include What You Use sanitizer" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
set(USE_LINKER "" CACHE STRING "Path to a custom linker program")
|
||||
option(ENABLE_UBSAN "Enable the Clang Undefined Behavior Sanitizer" FALSE)
|
||||
option(ENABLE_ASAN "Enable the Clang Address Sanitizer" FALSE)
|
||||
option(ENABLE_TSAN "Enable the Clang Thread Sanitizer" FALSE)
|
||||
option(ENABLE_COVERAGE "Enable Code Coverage" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enable debug assertions" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enable GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_STRICT_WERROR "Enable stricter -Werror" FALSE)
|
||||
option(ENABLE_WERROR "Enable -Werror" FALSE)
|
||||
option(ENABLE_FEX_ALLOCATOR "Enable allocator for FEX" TRUE)
|
||||
option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enable jemalloc glibc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enable FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enable time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Use LLVM's libc++ instead of the GNU libstdc++" FALSE)
|
||||
option(ENABLE_CCACHE "Enable ccache for build caching" TRUE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Use the VIXL simulator for emulation (only useful for CI testing)" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enable debug disassembler output with VIXL" FALSE)
|
||||
option(ENABLE_ZYDIS "Enable x86/x86-64 guest disassembler output with Zydis" FALSE)
|
||||
option(USE_LEGACY_BINFMTMISC "Use legacy method of setting up binfmt_misc" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enable FEXCore's timeline profiling capabilities" FALSE)
|
||||
set(FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for FEXCore's profiler")
|
||||
set_property(CACHE FEXCORE_PROFILER_BACKEND PROPERTY STRINGS gpuvis tracy)
|
||||
set(USE_LINKER "" CACHE STRING "Allow overriding the linker path directly")
|
||||
option(ENABLE_UBSAN "Enables Clang UBSAN" FALSE)
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_COVERAGE "Enables Coverage" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enables jemalloc glibc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Enable use of VIXL simulator for emulation (only useful for CI testing)" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for the FEXCore profiler (gpuvis, tracy)")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
option(USE_PDB_DEBUGINFO "Build debug info in PDB format" FALSE)
|
||||
option(BUILD_STEAM_SUPPORT "Enable Steam integration" FALSE)
|
||||
|
||||
set(X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set(X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set(X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
|
||||
set(DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
|
||||
set(HOSTLIBS_DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
|
||||
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
|
||||
set (DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
|
||||
set (HOSTLIBS_DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
|
||||
if (NOT DATA_DIRECTORY)
|
||||
set(DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
|
||||
endif()
|
||||
|
||||
include(GNUInstallDirs)
|
||||
@@ -52,95 +48,43 @@ if (NOT HOSTLIBS_DATA_DIRECTORY)
|
||||
set(HOSTLIBS_DATA_DIRECTORY "${CMAKE_INSTALL_FULL_LIBDIR}/fex-emu")
|
||||
endif()
|
||||
|
||||
## Platform Checks ##
|
||||
# Only 64-bit Linux and Windows are supported
|
||||
|
||||
# NB: SIZEOF_VOID_P is in bytes, not bits
|
||||
# On 32-bit systems this is set to 4
|
||||
if (NOT CMAKE_SIZEOF_VOID_P EQUAL 8)
|
||||
message(FATAL_ERROR "Unsupported pointer size ${CMAKE_SIZEOF_VOID_P}."
|
||||
" FEX only supports 64-bit (8-byte pointer) systems."
|
||||
" If you believe this is in error, file an issue.")
|
||||
elseif (NOT (WIN32 OR CMAKE_SYSTEM_NAME STREQUAL "Linux"))
|
||||
message(FATAL_ERROR "Unsupported system type ${CMAKE_SYSTEM_NAME}."
|
||||
" FEX only supports Linux and Windows."
|
||||
" If you believe this is in error, file an issue.")
|
||||
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
|
||||
if (NOT CONTAINS_MINGW EQUAL -1)
|
||||
message (STATUS "Mingw build")
|
||||
set (MINGW_BUILD TRUE)
|
||||
set (ENABLE_JEMALLOC TRUE)
|
||||
set (ENABLE_JEMALLOC_GLIBC_ALLOC FALSE)
|
||||
endif()
|
||||
|
||||
## Compiler Checks ##
|
||||
# GCC and MSVC are unsupported
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
message(FATAL_ERROR "FEX doesn't support GCC! Use Clang instead.")
|
||||
elseif (MSVC)
|
||||
message(FATAL_ERROR "FEX doesn't support MSVC! Use Clang on MinGW instead.")
|
||||
elseif (MINGW)
|
||||
message(STATUS "Building for MinGW")
|
||||
set(ENABLE_FEX_ALLOCATOR TRUE)
|
||||
set(ENABLE_JEMALLOC_GLIBC_ALLOC FALSE)
|
||||
else ()
|
||||
message(STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
set(CLANG_MINIMUM_VERSION 13.0)
|
||||
if (NOT MINGW_BUILD)
|
||||
message (STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
set (CLANG_MINIMUM_VERSION 13.0)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_LESS ${CLANG_MINIMUM_VERSION})
|
||||
message(FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
message (FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
## Architecture Handling ##
|
||||
string(TOLOWER ${CMAKE_SYSTEM_PROCESSOR} processor)
|
||||
if (processor MATCHES "x86|amd64")
|
||||
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
|
||||
if (NOT ENABLE_X86_HOST_DEBUG)
|
||||
message(FATAL_ERROR
|
||||
" FEX doesn't support compiling for x86-64 hosts!"
|
||||
" This is /only/ a supported configuration for FEX CI and nothing else!")
|
||||
else()
|
||||
message(STATUS "x86_64 debug build")
|
||||
endif()
|
||||
|
||||
set(ARCHITECTURE_x86_64 1)
|
||||
add_compile_definitions(ARCHITECTURE_x86_64=1)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
elseif (processor MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(ARCHITECTURE_arm64 1)
|
||||
add_compile_definitions(ARCHITECTURE_arm64=1)
|
||||
|
||||
# arm64ec needs to define both arm64 and arm64ec
|
||||
if (processor MATCHES "^arm64ec")
|
||||
set(ARCHITECTURE_arm64ec 1)
|
||||
add_compile_definitions(ARCHITECTURE_arm64ec=1)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (NOT (ARCHITECTURE_arm64 OR ARCHITECTURE_arm64ec OR ARCHITECTURE_x86_64))
|
||||
message(FATAL_ERROR "Unsupported processor type ${processor}."
|
||||
" If you believe this is in error, file an issue.")
|
||||
endif()
|
||||
|
||||
if (BUILD_STEAM_SUPPORT)
|
||||
add_compile_definitions(FEX_STEAM_SUPPORT=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_compile_definitions(ENABLE_FEXCORE_PROFILER=1)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
|
||||
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
|
||||
add_compile_definitions(FEXCORE_PROFILER_BACKEND=1)
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
|
||||
elseif (FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
add_compile_definitions(FEXCORE_PROFILER_BACKEND=2)
|
||||
add_compile_definitions(TRACY_ENABLE=1)
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=2)
|
||||
add_definitions(-DTRACY_ENABLE=1)
|
||||
# Required so that Tracy will only start in the selected guest application
|
||||
add_compile_definitions(TRACY_MANUAL_LIFETIME=1)
|
||||
add_compile_definitions(TRACY_DELAYED_INIT=1)
|
||||
add_definitions(-DTRACY_MANUAL_LIFETIME=1)
|
||||
add_definitions(-DTRACY_DELAYED_INIT=1)
|
||||
# This interferes with FEX's signal handling
|
||||
add_compile_definitions(TRACY_NO_CRASH_HANDLER=1)
|
||||
add_definitions(-DTRACY_NO_CRASH_HANDLER=1)
|
||||
# Tracy can gather call stack samples in regular intervals, but this
|
||||
# isn't useful for us since it would usually sample opaque JIT code
|
||||
add_compile_definitions(TRACY_NO_SAMPLING=1)
|
||||
add_definitions(-DTRACY_NO_SAMPLING=1)
|
||||
# This pulls in libbacktrace which allocators in global constructors (before FEX can set up its allocator hooks)
|
||||
add_compile_definitions(TRACY_NO_CALLSTACK=1)
|
||||
if (MINGW)
|
||||
message(FATAL_ERROR "Tracy profiler not supported on MinGW")
|
||||
add_definitions(-DTRACY_NO_CALLSTACK=1)
|
||||
if (MINGW_BUILD)
|
||||
message(FATAL_ERROR "Tracy profiler not supported")
|
||||
endif()
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
|
||||
@@ -152,7 +96,7 @@ if (ENABLE_JEMALLOC_GLIBC_ALLOC AND ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
|
||||
endif()
|
||||
|
||||
if (ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
|
||||
add_compile_definitions(GLIBC_ALLOCATOR_FAULT=1)
|
||||
add_definitions(-DGLIBC_ALLOCATOR_FAULT=1)
|
||||
endif()
|
||||
|
||||
# uninstall target
|
||||
@@ -167,17 +111,9 @@ if(NOT TARGET uninstall)
|
||||
endif()
|
||||
|
||||
# These options are meant for package management
|
||||
set(TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
|
||||
set(TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
set(OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version")
|
||||
set(OVERRIDE_HASH "detect" CACHE STRING "Override the FEX git hash")
|
||||
|
||||
get_property(IS_MULTI_CONFIG GLOBAL PROPERTY GENERATOR_IS_MULTI_CONFIG)
|
||||
if (NOT IS_MULTI_CONFIG AND NOT CMAKE_BUILD_TYPE)
|
||||
set(CMAKE_BUILD_TYPE Release
|
||||
CACHE STRING "Choose the type of build." FORCE)
|
||||
message(STATUS "No build type set, defaulting to a Release build")
|
||||
endif()
|
||||
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
|
||||
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
set (OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version in the format of <MMYY>{.<REV>}")
|
||||
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
|
||||
if (CMAKE_BUILD_TYPE MATCHES "DEBUG")
|
||||
@@ -186,18 +122,14 @@ endif()
|
||||
|
||||
if (ENABLE_ASSERTIONS)
|
||||
message(STATUS "Assertions enabled")
|
||||
add_compile_definitions(ASSERTIONS_ENABLED=1)
|
||||
add_definitions(-DASSERTIONS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_GDB_SYMBOLS)
|
||||
message(STATUS "GDBSymbols support enabled")
|
||||
add_compile_definitions(GDB_SYMBOLS_ENABLED=1)
|
||||
add_definitions(-DGDB_SYMBOLS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
add_compile_definitions(_LARGEFILE64_SOURCE)
|
||||
if (WIN32)
|
||||
add_compile_definitions(UNICODE _UNICODE)
|
||||
endif()
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
@@ -208,7 +140,33 @@ cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
|
||||
include(CheckPIESupported)
|
||||
check_pie_supported()
|
||||
|
||||
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION ${ENABLE_LTO})
|
||||
if (ENABLE_LTO)
|
||||
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION TRUE)
|
||||
else()
|
||||
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION FALSE)
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
|
||||
if (NOT ENABLE_X86_HOST_DEBUG)
|
||||
message(FATAL_ERROR
|
||||
" FEX-Emu doesn't support compiling for x86-64 hosts!"
|
||||
" This is /only/ a supported configuration for FEX CI and nothing else!")
|
||||
endif()
|
||||
set(_M_X86_64 1)
|
||||
add_definitions(-D_M_X86_64=1)
|
||||
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(_M_ARM_64 1)
|
||||
add_definitions(-D_M_ARM_64=1)
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
|
||||
set(_M_ARM_64EC 1)
|
||||
add_definitions(-D_M_ARM_64EC=1)
|
||||
endif()
|
||||
|
||||
include(CheckCXXSourceCompiles)
|
||||
set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
|
||||
@@ -224,47 +182,30 @@ check_cxx_source_compiles(
|
||||
HAS_CLANG_PRESERVE_ALL)
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
if (HAS_CLANG_PRESERVE_ALL)
|
||||
if (MINGW)
|
||||
if (MINGW_BUILD)
|
||||
message(STATUS "Ignoring broken clang::preserve_all support")
|
||||
set(HAS_CLANG_PRESERVE_ALL FALSE)
|
||||
else()
|
||||
message(STATUS "Has clang::preserve_all")
|
||||
endif()
|
||||
endif()
|
||||
endif ()
|
||||
|
||||
if (ARCHITECTURE_arm64 AND HAS_CLANG_PRESERVE_ALL)
|
||||
add_compile_definitions("FEX_PRESERVE_ALL_ATTR=__attribute__((preserve_all))" "FEX_HAS_PRESERVE_ALL_ATTR=1")
|
||||
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
|
||||
add_definitions("-DFEX_PRESERVE_ALL_ATTR=__attribute__((preserve_all))" "-DFEX_HAS_PRESERVE_ALL_ATTR=1")
|
||||
else()
|
||||
add_compile_definitions("FEX_PRESERVE_ALL_ATTR=" "FEX_HAS_PRESERVE_ALL_ATTR=0")
|
||||
add_definitions("-DFEX_PRESERVE_ALL_ATTR=" "-DFEX_HAS_PRESERVE_ALL_ATTR=0")
|
||||
endif()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#define _GNU_SOURCE
|
||||
#include <errno.h>
|
||||
int main() {
|
||||
return program_invocation_name == nullptr;
|
||||
}"
|
||||
HAS_PROGRAM_INVOCATION_NAME)
|
||||
add_compile_definitions("HAS_PROGRAM_INVOCATION_NAME=${HAS_PROGRAM_INVOCATION_NAME}")
|
||||
|
||||
if (ENABLE_VIXL_SIMULATOR)
|
||||
# We can run the simulator on both x86-64 or AArch64 hosts
|
||||
add_compile_definitions(VIXL_SIMULATOR=1 VIXL_INCLUDE_SIMULATOR_AARCH64=1)
|
||||
add_definitions(-DVIXL_SIMULATOR=1 -DVIXL_INCLUDE_SIMULATOR_AARCH64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_CCACHE)
|
||||
find_program(CCACHE_PROGRAM ccache)
|
||||
if(CCACHE_PROGRAM)
|
||||
execute_process(COMMAND "${CCACHE_PROGRAM}" --print-version
|
||||
OUTPUT_VARIABLE CCACHE_VERSION OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
message(STATUS "Enabling ccache ${CCACHE_VERSION}")
|
||||
if (CCACHE_VERSION VERSION_GREATER_EQUAL "4.8")
|
||||
# Set sloppiness to enable caching even for files that use __DATE__/__TIME__ macros
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM} sloppiness=time_macros")
|
||||
else()
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
|
||||
endif()
|
||||
message(STATUS "CCache enabled")
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -278,7 +219,7 @@ if (ENABLE_COMPILE_TIME_TRACE)
|
||||
link_libraries(-ftime-trace)
|
||||
endif()
|
||||
|
||||
set(PTHREAD_LIB pthread)
|
||||
set (PTHREAD_LIB pthread)
|
||||
|
||||
if (USE_LINKER)
|
||||
message(STATUS "Overriding linker to: ${USE_LINKER}")
|
||||
@@ -293,7 +234,7 @@ endif()
|
||||
|
||||
if (NOT ENABLE_OFFLINE_TELEMETRY)
|
||||
# Disable FEX offline telemetry entirely if asked
|
||||
add_compile_definitions(FEX_DISABLE_TELEMETRY=1)
|
||||
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_UBSAN)
|
||||
@@ -304,13 +245,13 @@ if (ENABLE_UBSAN)
|
||||
# that are regularly access unaligned.
|
||||
# function: syscalls cast function pointers to void (*)(unsigned long...), causing warnings
|
||||
# related to this access.
|
||||
add_compile_definitions(ENABLE_UBSAN=1)
|
||||
add_definitions(-DENABLE_UBSAN=1)
|
||||
add_compile_options(-fno-omit-frame-pointer -fsanitize=undefined -fno-sanitize=alignment -fno-sanitize=function -fno-sanitize-recover=undefined)
|
||||
link_libraries(-fno-omit-frame-pointer -fsanitize=undefined -fno-sanitize=alignment -fno-sanitize=function -fno-sanitize-recover=undefined)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
add_compile_definitions(ENABLE_ASAN=1)
|
||||
add_definitions(-DENABLE_ASAN=1)
|
||||
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
|
||||
link_libraries(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
|
||||
endif()
|
||||
@@ -330,20 +271,20 @@ if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
# Required for thunks to work.
|
||||
# All host native libraries will use this allocator, while *most* other FEX internal allocations will use the other jemalloc allocator.
|
||||
add_subdirectory(External/jemalloc_glibc/)
|
||||
elseif (NOT MINGW)
|
||||
message(STATUS
|
||||
elseif (NOT MINGW_BUILD)
|
||||
message (STATUS
|
||||
" jemalloc glibc allocator disabled!\n"
|
||||
" This is not a recommended configuration!\n"
|
||||
" This will very explicitly break thunk execution!\n"
|
||||
" Use at your own risk!")
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEX_ALLOCATOR)
|
||||
# The rpmalloc subproject that all FEXCore fextl objects allocate through.
|
||||
add_subdirectory(External/rpmalloc/)
|
||||
elseif (NOT MINGW)
|
||||
if (ENABLE_JEMALLOC)
|
||||
# The jemalloc subproject that all FEXCore fextl objects allocate through.
|
||||
add_subdirectory(External/jemalloc/)
|
||||
elseif (NOT MINGW_BUILD)
|
||||
message (STATUS
|
||||
" FEX allocator is disabled!\n"
|
||||
" jemalloc disabled!\n"
|
||||
" This is not a recommended configuration!\n"
|
||||
" This will very explicitly break 32-bit application execution!\n"
|
||||
" Use at your own risk!")
|
||||
@@ -354,64 +295,45 @@ if (USE_PDB_DEBUGINFO)
|
||||
add_link_options(-g -Wl,--pdb=)
|
||||
endif()
|
||||
|
||||
set(CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
set(CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
|
||||
set(CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
set(CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
|
||||
|
||||
## Modules ##
|
||||
list(APPEND CMAKE_MODULE_PATH ${CMAKE_SOURCE_DIR}/Data/CMake/)
|
||||
|
||||
include(LinkerGC)
|
||||
|
||||
## Externals ##
|
||||
|
||||
find_package(unordered_dense QUIET CONFIG)
|
||||
if (NOT unordered_dense_FOUND)
|
||||
add_subdirectory(External/unordered_dense)
|
||||
endif()
|
||||
include_directories(External/robin-map/include/)
|
||||
|
||||
include(CTest)
|
||||
if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
add_subdirectory(External/vixl/)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ZYDIS)
|
||||
find_package(Zycore 1.5 MODULE QUIET)
|
||||
find_package(Zydis 4.0 MODULE QUIET)
|
||||
|
||||
if (TARGET Zydis::Zydis AND TARGET Zycore::Zycore)
|
||||
message(STATUS "Using system Zydis")
|
||||
else()
|
||||
set(ZYDIS_BUILD_TOOLS OFF CACHE BOOL "" FORCE)
|
||||
set(ZYDIS_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE)
|
||||
|
||||
message(STATUS "Using bundled Zydis")
|
||||
add_subdirectory(External/zydis/)
|
||||
endif()
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
add_subdirectory(External/tracy)
|
||||
endif()
|
||||
|
||||
find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
# This means we were attempted to get compiled with GCC
|
||||
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
|
||||
endif()
|
||||
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
|
||||
if (NOT CMAKE_CROSSCOMPILING)
|
||||
find_package(xxhash MODULE QUIET)
|
||||
endif()
|
||||
|
||||
if (NOT TARGET xxHash::xxhash)
|
||||
pkg_search_module(xxhash IMPORTED_TARGET xxhash libxxhash)
|
||||
if (TARGET PkgConfig::xxhash AND NOT CMAKE_CROSSCOMPILING)
|
||||
add_library(xxHash::xxhash ALIAS PkgConfig::xxhash)
|
||||
else()
|
||||
set(XXHASH_BUNDLED_MODE TRUE)
|
||||
set(XXHASH_BUILD_XXHSUM FALSE)
|
||||
add_subdirectory(External/xxhash/cmake_unofficial/)
|
||||
endif()
|
||||
|
||||
add_compile_options(-Wno-trigraphs)
|
||||
add_compile_definitions(GLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTING)
|
||||
find_package(Catch2 3 QUIET)
|
||||
@@ -428,16 +350,11 @@ else ()
|
||||
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
|
||||
endif()
|
||||
|
||||
if (MINGW)
|
||||
find_package(fmt QUIET)
|
||||
if (NOT fmt_FOUND)
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
else()
|
||||
find_package(fmt QUIET)
|
||||
if (NOT fmt_FOUND)
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
find_package(range-v3 QUIET)
|
||||
@@ -447,6 +364,7 @@ if (NOT range-v3_FOUND)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/tiny-json/)
|
||||
include_directories(External/tiny-json/)
|
||||
|
||||
include_directories(Source/)
|
||||
include_directories("${CMAKE_BINARY_DIR}/Source/")
|
||||
@@ -470,11 +388,6 @@ if(ENUM_ENUM_WARNING)
|
||||
add_compile_options(-Wno-deprecated-enum-enum-conversion)
|
||||
endif()
|
||||
|
||||
# GCC enables -Wchanges-meaning by default and treats some cases as an error
|
||||
if(CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
add_compile_options(-Wno-error=changes-meaning)
|
||||
endif()
|
||||
|
||||
if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
add_compile_options(-Werror)
|
||||
if (NOT ENABLE_STRICT_WERROR)
|
||||
@@ -485,24 +398,16 @@ endif()
|
||||
|
||||
set(FEX_TUNE_COMPILE_FLAGS)
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
set(TUNE_ARCH_STRING "${TUNE_ARCH}")
|
||||
if(ARCHITECTURE_arm64)
|
||||
set(TUNE_ARCH_STRING "${TUNE_ARCH}+crc")
|
||||
endif()
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH_STRING}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH_STRING}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH_STRING}' but the compiler doesn't support this")
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
endif()
|
||||
elseif(ARCHITECTURE_arm64)
|
||||
# Need to always append crc
|
||||
check_cxx_compiler_flag("-march=armv8-a+crc" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=armv8-a+crc")
|
||||
endif()
|
||||
|
||||
if (TUNE_CPU STREQUAL "native")
|
||||
if(ARCHITECTURE_arm64)
|
||||
if(_M_ARM_64)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
|
||||
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
|
||||
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
|
||||
@@ -543,54 +448,8 @@ elseif (NOT TUNE_CPU STREQUAL "none")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
set(GIT_DESCRIBE_STRING "FEX-Unknown")
|
||||
|
||||
if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
find_package(Git)
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
endif()
|
||||
else()
|
||||
set(GIT_DESCRIBE_STRING "${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
set(GIT_HASH "Unknown")
|
||||
|
||||
if (OVERRIDE_HASH STREQUAL "detect")
|
||||
find_package(Git)
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
endif()
|
||||
else()
|
||||
set(GIT_HASH "${OVERRIDE_HASH}")
|
||||
endif()
|
||||
|
||||
message(STATUS "FEX version: ${GIT_DESCRIBE_STRING}")
|
||||
message(STATUS "FEX commit: ${GIT_HASH}")
|
||||
|
||||
# Prepends 0x to every two-character sequence in the hash,
|
||||
# OR the final character of the hash, to plumb it for C++ usage. e.g.:
|
||||
# -DOVERRIDE_HASH=123456aa => 0x12, 0x34, 0x56, 0xaa,
|
||||
# -DOVERRIDE_HASH=12345678a => 0x12, 0x34, 0x56, 0x78, 0xa,
|
||||
string(REGEX
|
||||
REPLACE "(..|.$)" "0x\\1, "
|
||||
GIT_HASH_ARRAY "${GIT_HASH}")
|
||||
|
||||
if (ENABLE_IWYU)
|
||||
find_program(IWYU_EXE
|
||||
NAMES iwyu include-what-you-use)
|
||||
find_program(IWYU_EXE "iwyu")
|
||||
if (IWYU_EXE)
|
||||
message(STATUS "IWYU enabled")
|
||||
set(CMAKE_CXX_INCLUDE_WHAT_YOU_USE "${IWYU_EXE}")
|
||||
@@ -602,7 +461,7 @@ add_compile_options(-Wall)
|
||||
if (BUILD_TESTING)
|
||||
message(STATUS "Unit tests are enabled")
|
||||
|
||||
set(TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
|
||||
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
|
||||
if (TEST_JOB_COUNT)
|
||||
message(STATUS "Running tests with ${TEST_JOB_COUNT} jobs")
|
||||
elseif(CMAKE_VERSION VERSION_LESS "3.29")
|
||||
@@ -617,16 +476,13 @@ add_subdirectory(FEXHeaderUtils/)
|
||||
add_subdirectory(CodeEmitter/)
|
||||
add_subdirectory(FEXCore/)
|
||||
|
||||
if (ARCHITECTURE_arm64 AND NOT MINGW AND NOT BUILD_STEAM_SUPPORT)
|
||||
if (_M_ARM_64 AND NOT MINGW_BUILD)
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(Source/)
|
||||
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
endif()
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
# Install the ThunksDB file
|
||||
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
|
||||
@@ -643,7 +499,7 @@ if (BUILD_TESTING)
|
||||
endif()
|
||||
|
||||
if (BUILD_THUNKS)
|
||||
set(FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
|
||||
set (FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
|
||||
add_subdirectory(ThunkLibs/Generator)
|
||||
|
||||
# Thunk targets for both host libraries and IDE integration
|
||||
@@ -670,7 +526,8 @@ if (BUILD_THUNKS)
|
||||
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen)
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
ExternalProject_Add(guest-libs-32
|
||||
PREFIX guest-libs-32
|
||||
@@ -688,36 +545,65 @@ if (BUILD_THUNKS)
|
||||
"-DX86_DEV_ROOTFS=${X86_DEV_ROOTFS}"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen)
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "message(\"-- Installing: guest-libs\")"
|
||||
CODE "MESSAGE(\"-- Installing: guest-libs\")"
|
||||
CODE "
|
||||
execute_process(COMMAND ${CMAKE_COMMAND} --build . --target install
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest)"
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)"
|
||||
DEPENDS guest-libs
|
||||
COMPONENT Runtime)
|
||||
COMPONENT Runtime
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "message(\"-- Installing: guest-libs-32\")"
|
||||
CODE "MESSAGE(\"-- Installing: guest-libs-32\")"
|
||||
CODE "
|
||||
execute_process(COMMAND ${CMAKE_COMMAND} --build . --target install
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32)"
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)"
|
||||
DEPENDS guest-libs-32
|
||||
COMPONENT Runtime)
|
||||
COMPONENT Runtime
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs
|
||||
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest)
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs-32
|
||||
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32)
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_guest-libs)
|
||||
add_dependencies(uninstall uninstall_guest-libs-32)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW AND BUILD_STEAM_SUPPORT)
|
||||
add_subdirectory(Source/Steam/)
|
||||
set(FEX_VERSION_MAJOR "0")
|
||||
set(FEX_VERSION_MINOR "0")
|
||||
set(FEX_VERSION_PATCH "0")
|
||||
|
||||
if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
find_package(Git)
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
RESULT_VARIABLE GIT_ERROR
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
|
||||
if (NOT ${GIT_ERROR} EQUAL 0)
|
||||
# Likely built in a way that doesn't have tags
|
||||
# Setup a version tag that is unknown
|
||||
set(GIT_DESCRIBE_STRING "FEX-0000")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
@@ -0,0 +1,132 @@
|
||||
{
|
||||
"environments": [
|
||||
{
|
||||
"BuildPath": "${projectDir}\\out\\build\\${name}",
|
||||
"InstallPath": "${projectDir}\\out\\install\\${name}",
|
||||
"clangcl": "clang-cl.exe",
|
||||
"cc": "clang",
|
||||
"cxx": "clang++"
|
||||
}
|
||||
],
|
||||
"configurations": [
|
||||
{
|
||||
"name": "WSL-Clang-Debug",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "Debug",
|
||||
"buildRoot": "${env.BuildPath}",
|
||||
"installRoot": "${env.InstallPath}",
|
||||
"cmakeExecutable": "/usr/bin/cmake",
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"wslPath": "${defaultWSLPath}",
|
||||
"inheritEnvironments": [ "linux_clang_x64" ],
|
||||
"addressSanitizerRuntimeFlags": "detect_leaks=0",
|
||||
"variables": [
|
||||
{
|
||||
"name": "WSL",
|
||||
"value": "TRUE",
|
||||
"type": "BOOL"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "WSL-Clang-Release",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "RelWithDebInfo",
|
||||
"buildRoot": "${env.BuildPath}",
|
||||
"installRoot": "${env.InstallPath}",
|
||||
"cmakeExecutable": "/usr/bin/cmake",
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"wslPath": "${defaultWSLPath}",
|
||||
"inheritEnvironments": [ "linux_clang_x64" ],
|
||||
"addressSanitizerRuntimeFlags": "detect_leaks=0",
|
||||
"variables": [
|
||||
{
|
||||
"name": "WSL",
|
||||
"value": "TRUE",
|
||||
"type": "BOOL"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "x86-Clang-Cross-Debug",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "Debug",
|
||||
"buildRoot": "${env.BuildPath}",
|
||||
"installRoot": "${env.InstallPath}",
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"inheritEnvironments": [ "clang_cl_x86" ],
|
||||
"variables": [
|
||||
{
|
||||
"name": "CMAKE_C_COMPILER",
|
||||
"value": "${env.cc}",
|
||||
"type": "STRING"
|
||||
},
|
||||
{
|
||||
"name": "CMAKE_CXX_COMPILER",
|
||||
"value": "${env.cxx}",
|
||||
"type": "STRING"
|
||||
},
|
||||
{
|
||||
"name": "CMAKE_SYSROOT",
|
||||
"value": "${env.fexsysroot}",
|
||||
"type": "STRING"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "x64-Clang-Cross-Release",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "RelWithDebInfo",
|
||||
"buildRoot": "${env.BuildPath}",
|
||||
"installRoot": "${env.InstallPath}",
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"inheritEnvironments": [ "clang_cl_x86" ],
|
||||
"variables": [
|
||||
{
|
||||
"name": "CMAKE_C_COMPILER",
|
||||
"value": "${env.cc}",
|
||||
"type": "STRING"
|
||||
},
|
||||
{
|
||||
"name": "CMAKE_CXX_COMPILER",
|
||||
"value": "${env.cxx}",
|
||||
"type": "STRING"
|
||||
},
|
||||
{
|
||||
"name": "CMAKE_SYSROOT",
|
||||
"value": "${env.fexsysroot}",
|
||||
"type": "STRING"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Linux-Clang-Remote-Debug",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "Debug",
|
||||
"cmakeExecutable": "/usr/bin/cmake",
|
||||
"remoteCopySourcesExclusionList": [ ".vs", ".vscode", ".git", ".github", "build", "out", "bin" ],
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"inheritEnvironments": [ "linux_clang_x64" ],
|
||||
"remoteMachineName": "${env.fexremote}",
|
||||
"remoteCMakeListsRoot": "$HOME/projects/.vs/${projectDirName}/src",
|
||||
"remoteBuildRoot": "$HOME/projects/.vs/${projectDirName}/build/${name}",
|
||||
"remoteInstallRoot": "$HOME/projects/.vs/${projectDirName}/install/${name}",
|
||||
"remoteCopySources": true,
|
||||
"rsyncCommandArgs": "-t --delete --delete-excluded",
|
||||
"remoteCopyBuildOutput": false,
|
||||
"remoteCopySourcesMethod": "rsync",
|
||||
"addressSanitizerRuntimeFlags": "detect_leaks=0",
|
||||
"variables": []
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1 +0,0 @@
|
||||
No AI/ML/LLM/etc code contributions.
|
||||
@@ -38,7 +38,9 @@ public:
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
if (IsADRRange(Imm)) {
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
|
||||
if (IsADRRange(Imm)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -71,8 +73,9 @@ public:
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) {
|
||||
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -100,22 +103,16 @@ public:
|
||||
}
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
const auto SLocation = reinterpret_cast<int64_t>(Label->Location);
|
||||
const auto ULocation = std::bit_cast<uint64_t>(SLocation);
|
||||
|
||||
const int64_t Imm = SLocation - (GetCursorAddress<int64_t>());
|
||||
const auto UImm = std::bit_cast<uint64_t>(Imm);
|
||||
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
|
||||
if (IsADRRange(Imm)) {
|
||||
// If the range is in ADR range then we can just use ADR.
|
||||
return adr(rd, Label);
|
||||
}
|
||||
if (IsADRPRange(Imm)) {
|
||||
const int64_t ADRPImm = (SLocation & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
} else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
|
||||
// If the range is in the ADRP range then we can use ADRP.
|
||||
const bool NeedsOffset = !IsADRPAligned(ULocation);
|
||||
const uint64_t AlignedOffset = ULocation & 0xFFFULL;
|
||||
bool NeedsOffset = !IsADRPAligned(reinterpret_cast<uint64_t>(Label->Location));
|
||||
uint64_t AlignedOffset = reinterpret_cast<uint64_t>(Label->Location) & 0xFFFULL;
|
||||
|
||||
// First emit ADRP
|
||||
adrp(rd, ADRPImm >> 12);
|
||||
@@ -128,19 +125,14 @@ public:
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Stinky path, we need to load the address as a sequence of movz+movk+movk
|
||||
movz(ARMEmitter::Size::i64Bit, rd, (UImm >> 32) & 0xFFFF, 32);
|
||||
movk(ARMEmitter::Size::i64Bit, rd, (UImm >> 16) & 0xFFFF, 16);
|
||||
movk(ARMEmitter::Size::i64Bit, rd, UImm & 0xFFFF);
|
||||
|
||||
return BranchEncodeSucceeded::Success;
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
|
||||
// Emit a register index and two nops. These will be backpatched.
|
||||
// Emit a register index and a nop. These will be backpatched.
|
||||
dc32(rd.Idx());
|
||||
nop();
|
||||
nop();
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
|
||||
@@ -22,7 +22,7 @@ public:
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -55,7 +55,7 @@ public:
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -116,7 +116,7 @@ public:
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded b(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -151,7 +151,7 @@ public:
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
|
||||
@@ -189,7 +189,7 @@ public:
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -227,7 +227,7 @@ public:
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -265,7 +265,7 @@ public:
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -301,8 +301,9 @@ public:
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
|
||||
@@ -53,7 +53,6 @@ public:
|
||||
if (!CurrentAlignment) {
|
||||
return;
|
||||
}
|
||||
std::memset(CurrentOffset, 0, Size - CurrentAlignment);
|
||||
CurrentOffset += Size - CurrentAlignment;
|
||||
}
|
||||
|
||||
|
||||
@@ -311,7 +311,7 @@ class ExtendedMemOperand final {
|
||||
public:
|
||||
ExtendedMemOperand(XRegister rn, XRegister rm = XReg::zr, ExtendedType Option = ExtendedType::LSL_64, uint32_t Shift = 0)
|
||||
: rn {rn}
|
||||
, MetaType {.Extended {
|
||||
, MetaType {.ExtendedType {
|
||||
.Header = {.MemType = TYPE_EXTENDED},
|
||||
.rm = rm,
|
||||
.Option = Option,
|
||||
@@ -340,7 +340,7 @@ public:
|
||||
Register rm;
|
||||
ExtendedType Option;
|
||||
uint32_t Shift;
|
||||
} Extended;
|
||||
} ExtendedType;
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
IndexType Index;
|
||||
@@ -586,10 +586,6 @@ concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegi
|
||||
template<typename T>
|
||||
concept IsQOrDRegister = std::is_same_v<T, QRegister> || std::is_same_v<T, DRegister>;
|
||||
|
||||
template<typename T>
|
||||
concept IsLabel = std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>;
|
||||
|
||||
enum class BranchEncodeSucceeded {
|
||||
Success,
|
||||
Failure,
|
||||
@@ -662,7 +658,7 @@ public:
|
||||
case ForwardLabel::InstType::ADR: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
if (!IsADRRange(Imm)) {
|
||||
if (!IsADRRange(Imm)) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -678,7 +674,7 @@ public:
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
|
||||
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) {
|
||||
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -695,7 +691,7 @@ public:
|
||||
case ForwardLabel::InstType::B: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) {
|
||||
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -711,7 +707,7 @@ public:
|
||||
case ForwardLabel::InstType::TEST_BRANCH: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) {
|
||||
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -728,7 +724,7 @@ public:
|
||||
case ForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) {
|
||||
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -741,44 +737,38 @@ public:
|
||||
break;
|
||||
}
|
||||
case ForwardLabel::InstType::LONG_ADDRESS_GEN: {
|
||||
const auto* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
const auto ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
const auto ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
const auto ImmInstThree = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[2]);
|
||||
const auto OriginalOffset = GetCursorOffset();
|
||||
uint32_t* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
auto OriginalOffset = GetCursorOffset();
|
||||
|
||||
const auto InstOffset = GetCursorOffsetFromAddress(Instructions);
|
||||
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
|
||||
SetCursorOffset(InstOffset);
|
||||
|
||||
// We encoded the destination register in to the first instruction space.
|
||||
// Read it back.
|
||||
ARMEmitter::Register DestReg(Instructions[0]);
|
||||
|
||||
if (IsADRRange(ImmInstThree)) {
|
||||
// If within ADR range from the third instruction, then we can emit NOP+NOP+ADR
|
||||
if (IsADRRange(ImmInstTwo)) {
|
||||
// If within ADR range from the second instruction, then we can emit NOP+ADR
|
||||
nop();
|
||||
nop();
|
||||
adr(DestReg, static_cast<uint32_t>(ImmInstThree) & 0x7FFF);
|
||||
} else if (IsADRPRange(ImmInstTwo)) {
|
||||
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
|
||||
} else if (IsADRPRange(ImmInstOne)) {
|
||||
|
||||
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
|
||||
// First check if we are in non-offset range for second instruction.
|
||||
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
|
||||
// We can emit nop + nop + adrp
|
||||
nop();
|
||||
nop();
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstThree >> 12) & 0x7FFF);
|
||||
} else {
|
||||
// Not aligned, need nop + adrp + add
|
||||
// We can emit nop + adrp
|
||||
nop();
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
|
||||
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstTwo & 0xFFF);
|
||||
} else {
|
||||
// Not aligned, need adrp + add
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
|
||||
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
|
||||
}
|
||||
} else {
|
||||
// Stinky path, we need to emit a movz+movk+movk sequence.
|
||||
movz(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 32) & 0x7FFF, 32);
|
||||
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 16) & 0xFFFF, 16);
|
||||
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne) & 0xFFFF);
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
SetCursorOffset(OriginalOffset);
|
||||
|
||||
@@ -3627,8 +3627,8 @@ public:
|
||||
|
||||
void strb(ARMEmitter::Register rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3650,8 +3650,8 @@ public:
|
||||
}
|
||||
void ldrb(ARMEmitter::Register rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3673,8 +3673,8 @@ public:
|
||||
}
|
||||
void ldrsb(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3696,8 +3696,8 @@ public:
|
||||
}
|
||||
void ldrsb(ARMEmitter::WRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3719,8 +3719,8 @@ public:
|
||||
}
|
||||
void strh(ARMEmitter::Register rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3742,8 +3742,8 @@ public:
|
||||
}
|
||||
void ldrh(ARMEmitter::Register rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3765,8 +3765,8 @@ public:
|
||||
}
|
||||
void ldrsh(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3788,8 +3788,8 @@ public:
|
||||
}
|
||||
void ldrsh(ARMEmitter::WRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3811,8 +3811,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::WRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3834,8 +3834,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::WRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3857,8 +3857,8 @@ public:
|
||||
}
|
||||
void ldrsw(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsw(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsw(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsw(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3880,8 +3880,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3903,8 +3903,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3926,8 +3926,8 @@ public:
|
||||
}
|
||||
void prfm(ARMEmitter::Prefetch prfop, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
prfm(prfop, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
prfm(prfop, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3946,9 +3946,9 @@ public:
|
||||
|
||||
void strb(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_A_FMT(MemSrc.MetaType.Extended.Shift == false, "Can't shift byte");
|
||||
strb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_A_FMT(MemSrc.MetaType.ExtendedType.Shift == false, "Can't shift byte");
|
||||
strb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3970,9 +3970,9 @@ public:
|
||||
}
|
||||
void ldrb(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_A_FMT(MemSrc.MetaType.Extended.Shift == false, "Can't shift byte");
|
||||
ldrb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_A_FMT(MemSrc.MetaType.ExtendedType.Shift == false, "Can't shift byte");
|
||||
ldrb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3994,8 +3994,8 @@ public:
|
||||
}
|
||||
void strh(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4017,8 +4017,8 @@ public:
|
||||
}
|
||||
void ldrh(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4040,8 +4040,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::SRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4063,8 +4063,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::SRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4086,8 +4086,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::DRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4109,8 +4109,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::DRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4132,8 +4132,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::QRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4155,8 +4155,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::QRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
|
||||
@@ -270,9 +270,6 @@ public:
|
||||
void fcvtxnt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFloatConvertOdd(0b00, 0b10, pg, zn, zd);
|
||||
}
|
||||
void bfcvtnt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFloatConvertOdd(0b10, 0b10, pg, zn, zd);
|
||||
}
|
||||
///< Size is destination size
|
||||
void fcvtnt(SubRegSize size, ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i32Bit || size == SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
@@ -295,6 +292,8 @@ public:
|
||||
SVEFloatConvertOdd(ConvertedSrcSize, ConvertedDestSize, pg, zn, zd);
|
||||
}
|
||||
|
||||
// XXX: BFCVTNT
|
||||
|
||||
// SVE2 floating-point pairwise operations
|
||||
void faddp(SubRegSize size, ZRegister zd, PRegisterMerge pg, ZRegister zn, ZRegister zm) {
|
||||
SVEFloatPairwiseArithmetic(0b000, size, pg, zd, zn, zm);
|
||||
@@ -2313,15 +2312,15 @@ public:
|
||||
|
||||
// SVE floating-point convert precision
|
||||
void fcvt(SubRegSize to, SubRegSize from, ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(to != from, "to and from sizes cannot be the same.");
|
||||
LOGMAN_THROW_A_FMT(to != SubRegSize::i8Bit && from != SubRegSize::i8Bit, "Can't use 8-bit element size");
|
||||
SVEFPConvertPrecision(to, from, zd, pg, zn);
|
||||
}
|
||||
void fcvtx(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFPConvertPrecision(SubRegSize::i32Bit, SubRegSize::i8Bit, zd, pg, zn);
|
||||
}
|
||||
void bfcvt(ZRegister zd, PRegisterMerge pg, ZRegister zn) {
|
||||
SVEFPConvertPrecision(SubRegSize::i32Bit, SubRegSize::i32Bit, zd, pg, zn);
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
uint32_t Instr = 0b0110'0101'0000'1010'1010'0000'0000'0000;
|
||||
Instr |= pg.Idx() << 10;
|
||||
Instr |= zn.Idx() << 5;
|
||||
Instr |= zd.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// SVE floating-point unary operations
|
||||
@@ -3848,19 +3847,14 @@ private:
|
||||
|
||||
void SVEFPConvertPrecision(SubRegSize to, SubRegSize from, ZRegister zd, PRegister pg, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
LOGMAN_THROW_A_FMT(to != SubRegSize::i128Bit && from != SubRegSize::i128Bit, "Can't use 128-bit element size");
|
||||
LOGMAN_THROW_A_FMT(to != from, "to and from sizes cannot be the same.");
|
||||
LOGMAN_THROW_A_FMT(to != SubRegSize::i8Bit && to != SubRegSize::i128Bit && from != SubRegSize::i8Bit && from != SubRegSize::i128Bit,
|
||||
"Can't use 8-bit or 128-bit element size");
|
||||
|
||||
// Encodings for the to and from sizes can get a little funky
|
||||
// depending on what is being converted to/from.
|
||||
const uint32_t op = [&] {
|
||||
switch (from) {
|
||||
case SubRegSize::i8Bit: {
|
||||
switch (to) {
|
||||
case SubRegSize::i32Bit: return 0x00020000U;
|
||||
default: return UINT32_MAX;
|
||||
}
|
||||
}
|
||||
|
||||
case SubRegSize::i16Bit: {
|
||||
switch (to) {
|
||||
case SubRegSize::i32Bit: return 0x00810000U;
|
||||
@@ -3872,7 +3866,6 @@ private:
|
||||
case SubRegSize::i32Bit: {
|
||||
switch (to) {
|
||||
case SubRegSize::i16Bit: return 0x00800000U;
|
||||
case SubRegSize::i32Bit: return 0x00820000U;
|
||||
case SubRegSize::i64Bit: return 0x00C30000U;
|
||||
default: return UINT32_MAX;
|
||||
}
|
||||
@@ -5132,7 +5125,7 @@ private:
|
||||
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
|
||||
[[nodiscard]]
|
||||
static bool IsValidFPValueForImm8(T value) {
|
||||
const uint64_t bits = std::bit_cast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
|
||||
|
||||
static constexpr std::array mantissa_masks {
|
||||
@@ -5178,7 +5171,7 @@ protected:
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = std::bit_cast<uint32_t>(value);
|
||||
const auto bits = FEXCore::BitCast<uint32_t>(value);
|
||||
const auto sign = (bits & 0x80000000) >> 24;
|
||||
const auto expb2 = (bits & 0x20000000) >> 23;
|
||||
const auto b5_to_0 = (bits >> 19) & 0x3F;
|
||||
@@ -5191,7 +5184,7 @@ protected:
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = std::bit_cast<uint64_t>(value);
|
||||
const auto bits = FEXCore::BitCast<uint64_t>(value);
|
||||
const auto sign = (bits & 0x80000000'00000000) >> 56;
|
||||
const auto expb2 = (bits & 0x20000000'00000000) >> 55;
|
||||
const auto b5_to_0 = (bits >> 48) & 0x3F;
|
||||
|
||||
@@ -15,10 +15,13 @@ foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
|
||||
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WLE)
|
||||
|
||||
# Configure it
|
||||
configure_file(${GEN_CONFIG_SRC} ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
|
||||
configure_file(
|
||||
${GEN_CONFIG_SRC}
|
||||
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
|
||||
|
||||
# Then install the configured json
|
||||
install(FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
@@ -1,23 +0,0 @@
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
if (CMAKE_CROSSCOMPILING)
|
||||
return()
|
||||
endif()
|
||||
|
||||
include(FindPackageHandleStandardArgs)
|
||||
|
||||
find_package(Zycore QUIET CONFIG)
|
||||
|
||||
if (Zycore_CONSIDERED_CONFIGS)
|
||||
find_package_handle_standard_args(Zycore CONFIG_MODE)
|
||||
else()
|
||||
find_package(PkgConfig QUIET)
|
||||
pkg_search_module(Zycore QUIET IMPORTED_TARGET zycore)
|
||||
find_package_handle_standard_args(Zycore
|
||||
REQUIRED_VARS zycore_LINK_LIBRARIES
|
||||
VERSION_VAR zycore_VERSION)
|
||||
|
||||
if (TARGET PkgConfig::zycore)
|
||||
add_library(Zycore::Zycore ALIAS PkgConfig::zycore)
|
||||
endif()
|
||||
endif()
|
||||
@@ -1,23 +0,0 @@
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
if (CMAKE_CROSSCOMPILING)
|
||||
return()
|
||||
endif()
|
||||
|
||||
include(FindPackageHandleStandardArgs)
|
||||
|
||||
find_package(Zydis QUIET CONFIG)
|
||||
|
||||
if (Zydis_CONSIDERED_CONFIGS)
|
||||
find_package_handle_standard_args(Zydis CONFIG_MODE)
|
||||
else()
|
||||
find_package(PkgConfig QUIET)
|
||||
pkg_search_module(Zydis QUIET IMPORTED_TARGET zydis)
|
||||
find_package_handle_standard_args(Zydis
|
||||
REQUIRED_VARS zydis_LINK_LIBRARIES
|
||||
VERSION_VAR zydis_VERSION)
|
||||
|
||||
if (TARGET PkgConfig::zydis)
|
||||
add_library(Zydis::Zydis ALIAS PkgConfig::zydis)
|
||||
endif()
|
||||
endif()
|
||||
@@ -1,18 +0,0 @@
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
include(FindPackageHandleStandardArgs)
|
||||
|
||||
find_package(PkgConfig QUIET)
|
||||
pkg_search_module(xxhash QUIET IMPORTED_TARGET xxhash libxxhash)
|
||||
find_package_handle_standard_args(xxhash
|
||||
REQUIRED_VARS xxhash_LINK_LIBRARIES
|
||||
VERSION_VAR xxhash_VERSION
|
||||
)
|
||||
|
||||
if (xxhash_FOUND AND NOT TARGET xxHash::xxhash)
|
||||
if (TARGET PkgConfig::xxhash)
|
||||
add_library(xxHash::xxhash ALIAS PkgConfig::xxhash)
|
||||
else()
|
||||
add_library(xxHash::xxhash ALIAS xxhash)
|
||||
endif()
|
||||
endif()
|
||||
@@ -1,15 +0,0 @@
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
# This applies some common linker options that reduce code size and linking time in Release mode. Namely:
|
||||
# --gc-sections: Linktime garbage collection, discards unused sections from the final output
|
||||
# --strip-all : Similar to running `strip`, discards the symbol table from the final output
|
||||
# --as-needed : Only includes libraries that are actually needed in the final output.
|
||||
|
||||
macro(LinkerGC target)
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
target_link_options(${target} PRIVATE
|
||||
"LINKER:--gc-sections"
|
||||
"LINKER:--strip-all"
|
||||
"LINKER:--as-needed")
|
||||
endif()
|
||||
endmacro()
|
||||
@@ -9,8 +9,8 @@ set(CMAKE_AR ${MINGW_TRIPLE}-ar)
|
||||
# Compile everything as static to avoid requiring the MinGW runtime libraries, force page aligned sections so that
|
||||
# debug symbols work correctly, and disable loop alignment to workaround an LLVM bug
|
||||
# (https://github.com/llvm/llvm-project/issues/47432)
|
||||
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-static -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-static -static-libgcc -static-libstdc++ -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_EXE_LINKER_FLAGS_INIT "-static -static-libgcc -static-libstdc++ -Wl,--file-alignment=4096,/mllvm:-align-loops=1")
|
||||
set(CMAKE_C_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
|
||||
set(CMAKE_CXX_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
|
||||
set(CMAKE_STANDARD_LIBRARIES "" CACHE STRING "" FORCE)
|
||||
|
||||
@@ -46,13 +46,6 @@
|
||||
"@PREFIX_LIB@/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"cuda": {
|
||||
"Library" : "libcuda-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libcuda.so",
|
||||
"@PREFIX_LIB@/libcuda.so.1"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3,10 +3,13 @@ function(GenBinFmt Name)
|
||||
get_filename_component(FMT_NAME ${Name} NAME_WE)
|
||||
|
||||
# Configure it
|
||||
configure_file(${Name} ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
|
||||
configure_file(
|
||||
${Name}
|
||||
${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
|
||||
|
||||
# Then install the configured binfmt
|
||||
install(FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/
|
||||
COMPONENT Runtime)
|
||||
endfunction()
|
||||
@@ -15,7 +18,8 @@ if (NOT USE_LEGACY_BINFMTMISC)
|
||||
configure_file(FEX-x86.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf)
|
||||
configure_file(FEX-x86_64.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf)
|
||||
|
||||
install(FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/
|
||||
COMPONENT Runtime)
|
||||
else()
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
|
||||
let
|
||||
toolchain = pkgs.fetchzip {
|
||||
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250920/llvm-mingw-20250920-ucrt-ubuntu-22.04-aarch64.tar.xz";
|
||||
sha256 = "sha256-LaojKjC8KzY+soW5u6eoDoXE3qtYk9Ejr7M3enTqRAE=";
|
||||
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250305/llvm-mingw-20250305-ucrt-ubuntu-20.04-aarch64.tar.xz";
|
||||
sha256 = "sha256-cA03/ab9O61eO9+S2JzIXD4V0HzTXK5/AYyxW2d73Po=";
|
||||
};
|
||||
|
||||
cmakeToolchainFile = pkgs.substitute {
|
||||
|
||||
Vendored
+1
-1
Submodule External/Catch2 updated: b3fb4b9fea...8ac8190e49.
Vendored
+3
-2
@@ -1,5 +1,5 @@
|
||||
|
||||
add_library(softfloat_3e STATIC
|
||||
set (SRCS
|
||||
# F80 support
|
||||
src/extF80_add.c
|
||||
src/extF80_div.c
|
||||
@@ -84,7 +84,7 @@ add_library(softfloat_3e STATIC
|
||||
src/s_normSubnormalF32Sig.c
|
||||
src/s_f32UIToCommonNaN.c)
|
||||
|
||||
if (ARCHITECTURE_arm64 AND HAS_CLANG_PRESERVE_ALL)
|
||||
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
|
||||
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=__attribute__((preserve_all));-DFEXCORE_HAS_PRESERVE_ALL_ATTR=1")
|
||||
else()
|
||||
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=;-DFEXCORE_HAS_PRESERVE_ALL_ATTR=0")
|
||||
@@ -92,6 +92,7 @@ endif()
|
||||
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ=1;-DINLINE=static inline;-DINLINE_LEVEL=4;-DSOFTFLOAT_FAST_INT64=1;-DSOFTFLOAT_FAST_DIV32TO16=1;-DSOFTFLOAT_FAST_DIV64TO32=1")
|
||||
|
||||
add_library(softfloat_3e STATIC ${SRCS})
|
||||
target_include_directories(softfloat_3e PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include/)
|
||||
target_include_directories(softfloat_3e PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include/SoftFloat-3e/)
|
||||
target_compile_definitions(softfloat_3e PUBLIC ${DEFINES})
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 450bd22322...cacef3039d.
Vendored
+2
-1
@@ -1,4 +1,4 @@
|
||||
add_library(cephes_128bit STATIC
|
||||
set(SRCS_128BIT
|
||||
src/128bit/Impl.cpp
|
||||
src/128bit/atanll.c
|
||||
src/128bit/constll.c
|
||||
@@ -11,6 +11,7 @@ add_library(cephes_128bit STATIC
|
||||
src/128bit/tanll.c)
|
||||
|
||||
# 128-bit library
|
||||
add_library(cephes_128bit STATIC ${SRCS_128BIT})
|
||||
target_link_libraries(cephes_128bit softfloat_3e)
|
||||
target_include_directories(cephes_128bit PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include/)
|
||||
target_compile_options(cephes_128bit PRIVATE -fno-builtin)
|
||||
+155
-252
@@ -1,37 +1,32 @@
|
||||
#
|
||||
# This file is autogenerated by pip-compile with Python 3.14
|
||||
# This file is autogenerated by pip-compile with Python 3.13
|
||||
# by the following command:
|
||||
#
|
||||
# pip-compile --generate-hashes --output-file=requirements_formatting.txt --strip-extras requirements_formatting.txt.in
|
||||
#
|
||||
black==26.3.1 \
|
||||
--hash=sha256:0126ae5b7c09957da2bdbd91a9ba1207453feada9e9fe51992848658c6c8e01c \
|
||||
--hash=sha256:0f76ff19ec5297dd8e66eb64deda23631e642c9393ab592826fd4bdc97a4bce7 \
|
||||
--hash=sha256:28ef38aee69e4b12fda8dba75e21f9b4f979b490c8ac0baa7cb505369ac9e1ff \
|
||||
--hash=sha256:2bd5aa94fc267d38bb21a70d7410a89f1a1d318841855f698746f8e7f51acd1b \
|
||||
--hash=sha256:2c50f5063a9641c7eed7795014ba37b0f5fa227f3d408b968936e24bc0566b07 \
|
||||
--hash=sha256:2d6bfaf7fd0993b420bed691f20f9492d53ce9a2bcccea4b797d34e947318a78 \
|
||||
--hash=sha256:41cd2012d35b47d589cb8a16faf8a32ef7a336f56356babd9fcf70939ad1897f \
|
||||
--hash=sha256:474c27574d6d7037c1bc875a81d9be0a9a4f9ee95e62800dab3cfaadbf75acd5 \
|
||||
--hash=sha256:5602bdb96d52d2d0672f24f6ffe5218795736dd34807fd0fd55ccd6bf206168b \
|
||||
--hash=sha256:5e9d0d86df21f2e1677cc4bd090cd0e446278bcbbe49bf3659c308c3e402843e \
|
||||
--hash=sha256:5ed0ca58586c8d9a487352a96b15272b7fa55d139fc8496b519e78023a8dab0a \
|
||||
--hash=sha256:6c54a4a82e291a1fee5137371ab488866b7c86a3305af4026bdd4dc78642e1ac \
|
||||
--hash=sha256:6e131579c243c98f35bce64a7e08e87fb2d610544754675d4a0e73a070a5aa3a \
|
||||
--hash=sha256:855822d90f884905362f602880ed8b5df1b7e3ee7d0db2502d4388a954cc8c54 \
|
||||
--hash=sha256:86a8b5035fce64f5dcd1b794cf8ec4d31fe458cf6ce3986a30deb434df82a1d2 \
|
||||
--hash=sha256:8a33d657f3276328ce00e4d37fe70361e1ec7614da5d7b6e78de5426cb56332f \
|
||||
--hash=sha256:92c0ec1f2cc149551a2b7b47efc32c866406b6891b0ee4625e95967c8f4acfb1 \
|
||||
--hash=sha256:9a5e9f45e5d5e1c5b5c29b3bd4265dcc90e8b92cf4534520896ed77f791f4da5 \
|
||||
--hash=sha256:afc622538b430aa4c8c853f7f63bc582b3b8030fd8c80b70fb5fa5b834e575c2 \
|
||||
--hash=sha256:b07fc0dab849d24a80a29cfab8d8a19187d1c4685d8a5e6385a5ce323c1f015f \
|
||||
--hash=sha256:b5e6f89631eb88a7302d416594a32faeee9fb8fb848290da9d0a5f2903519fc1 \
|
||||
--hash=sha256:bf9bf162ed91a26f1adba8efda0b573bc6924ec1408a52cc6f82cb73ec2b142c \
|
||||
--hash=sha256:c7e72339f841b5a237ff14f7d3880ddd0fc7f98a1199e8c4327f9a4f478c1839 \
|
||||
--hash=sha256:ddb113db38838eb9f043623ba274cfaf7d51d5b0c22ecb30afe58b1bb8322983 \
|
||||
--hash=sha256:dfdd51fc3e64ea4f35873d1b3fb25326773d55d2329ff8449139ebaad7357efb \
|
||||
--hash=sha256:f1cd08e99d2f9317292a311dfe578fd2a24b15dbce97792f9c4d752275c1fa56 \
|
||||
--hash=sha256:f89f2ab047c76a9c03f78d0d66ca519e389519902fa27e7a91117ef7611c0568
|
||||
black==25.1.0 \
|
||||
--hash=sha256:030b9759066a4ee5e5aca28c3c77f9c64789cdd4de8ac1df642c40b708be6171 \
|
||||
--hash=sha256:055e59b198df7ac0b7efca5ad7ff2516bca343276c466be72eb04a3bcc1f82d7 \
|
||||
--hash=sha256:0e519ecf93120f34243e6b0054db49c00a35f84f195d5bce7e9f5cfc578fc2da \
|
||||
--hash=sha256:172b1dbff09f86ce6f4eb8edf9dede08b1fce58ba194c87d7a4f1a5aa2f5b3c2 \
|
||||
--hash=sha256:1e2978f6df243b155ef5fa7e558a43037c3079093ed5d10fd84c43900f2d8ecc \
|
||||
--hash=sha256:33496d5cd1222ad73391352b4ae8da15253c5de89b93a80b3e2c8d9a19ec2666 \
|
||||
--hash=sha256:3b48735872ec535027d979e8dcb20bf4f70b5ac75a8ea99f127c106a7d7aba9f \
|
||||
--hash=sha256:4b60580e829091e6f9238c848ea6750efed72140b91b048770b64e74fe04908b \
|
||||
--hash=sha256:759e7ec1e050a15f89b770cefbf91ebee8917aac5c20483bc2d80a6c3a04df32 \
|
||||
--hash=sha256:8f0b18a02996a836cc9c9c78e5babec10930862827b1b724ddfe98ccf2f2fe4f \
|
||||
--hash=sha256:95e8176dae143ba9097f351d174fdaf0ccd29efb414b362ae3fd72bf0f710717 \
|
||||
--hash=sha256:96c1c7cd856bba8e20094e36e0f948718dc688dba4a9d78c3adde52b9e6c2299 \
|
||||
--hash=sha256:a1ee0a0c330f7b5130ce0caed9936a904793576ef4d2b98c40835d6a65afa6a0 \
|
||||
--hash=sha256:a22f402b410566e2d1c950708c77ebf5ebd5d0d88a6a2e87c86d9fb48afa0d18 \
|
||||
--hash=sha256:a39337598244de4bae26475f77dda852ea00a93bd4c728e09eacd827ec929df0 \
|
||||
--hash=sha256:afebb7098bfbc70037a053b91ae8437c3857482d3a690fefc03e9ff7aa9a5fd3 \
|
||||
--hash=sha256:bacabb307dca5ebaf9c118d2d2f6903da0d62c9faa82bd21a33eecc319559355 \
|
||||
--hash=sha256:bce2e264d59c91e52d8000d507eb20a9aca4a778731a08cfff7e5ac4a4bb7096 \
|
||||
--hash=sha256:d9e6827d563a2c820772b32ce8a42828dc6790f095f441beef18f96aa6f8294e \
|
||||
--hash=sha256:db8ea9917d6f8fc62abd90d944920d95e73c83a5ee3383493e35d271aca872e9 \
|
||||
--hash=sha256:ea0213189960bda9cf99be5b8c8ce66bb054af5e9e861249cd23471bd7b0b3ba \
|
||||
--hash=sha256:f3df5f1bf91d36002b0a75389ca8663510cf0531cca8aa5c1ef695b46d98655f
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# darker
|
||||
@@ -41,91 +36,71 @@ certifi==2025.7.14 \
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# requests
|
||||
cffi==2.0.0 \
|
||||
--hash=sha256:00bdf7acc5f795150faa6957054fbbca2439db2f775ce831222b66f192f03beb \
|
||||
--hash=sha256:07b271772c100085dd28b74fa0cd81c8fb1a3ba18b21e03d7c27f3436a10606b \
|
||||
--hash=sha256:087067fa8953339c723661eda6b54bc98c5625757ea62e95eb4898ad5e776e9f \
|
||||
--hash=sha256:0a1527a803f0a659de1af2e1fd700213caba79377e27e4693648c2923da066f9 \
|
||||
--hash=sha256:0cf2d91ecc3fcc0625c2c530fe004f82c110405f101548512cce44322fa8ac44 \
|
||||
--hash=sha256:0f6084a0ea23d05d20c3edcda20c3d006f9b6f3fefeac38f59262e10cef47ee2 \
|
||||
--hash=sha256:12873ca6cb9b0f0d3a0da705d6086fe911591737a59f28b7936bdfed27c0d47c \
|
||||
--hash=sha256:19f705ada2530c1167abacb171925dd886168931e0a7b78f5bffcae5c6b5be75 \
|
||||
--hash=sha256:1cd13c99ce269b3ed80b417dcd591415d3372bcac067009b6e0f59c7d4015e65 \
|
||||
--hash=sha256:1e3a615586f05fc4065a8b22b8152f0c1b00cdbc60596d187c2a74f9e3036e4e \
|
||||
--hash=sha256:1f72fb8906754ac8a2cc3f9f5aaa298070652a0ffae577e0ea9bd480dc3c931a \
|
||||
--hash=sha256:1fc9ea04857caf665289b7a75923f2c6ed559b8298a1b8c49e59f7dd95c8481e \
|
||||
--hash=sha256:203a48d1fb583fc7d78a4c6655692963b860a417c0528492a6bc21f1aaefab25 \
|
||||
--hash=sha256:2081580ebb843f759b9f617314a24ed5738c51d2aee65d31e02f6f7a2b97707a \
|
||||
--hash=sha256:21d1152871b019407d8ac3985f6775c079416c282e431a4da6afe7aefd2bccbe \
|
||||
--hash=sha256:24b6f81f1983e6df8db3adc38562c83f7d4a0c36162885ec7f7b77c7dcbec97b \
|
||||
--hash=sha256:256f80b80ca3853f90c21b23ee78cd008713787b1b1e93eae9f3d6a7134abd91 \
|
||||
--hash=sha256:28a3a209b96630bca57cce802da70c266eb08c6e97e5afd61a75611ee6c64592 \
|
||||
--hash=sha256:2c8f814d84194c9ea681642fd164267891702542f028a15fc97d4674b6206187 \
|
||||
--hash=sha256:2de9a304e27f7596cd03d16f1b7c72219bd944e99cc52b84d0145aefb07cbd3c \
|
||||
--hash=sha256:38100abb9d1b1435bc4cc340bb4489635dc2f0da7456590877030c9b3d40b0c1 \
|
||||
--hash=sha256:3925dd22fa2b7699ed2617149842d2e6adde22b262fcbfada50e3d195e4b3a94 \
|
||||
--hash=sha256:3e17ed538242334bf70832644a32a7aae3d83b57567f9fd60a26257e992b79ba \
|
||||
--hash=sha256:3e837e369566884707ddaf85fc1744b47575005c0a229de3327f8f9a20f4efeb \
|
||||
--hash=sha256:3f4d46d8b35698056ec29bca21546e1551a205058ae1a181d871e278b0b28165 \
|
||||
--hash=sha256:44d1b5909021139fe36001ae048dbdde8214afa20200eda0f64c068cac5d5529 \
|
||||
--hash=sha256:45d5e886156860dc35862657e1494b9bae8dfa63bf56796f2fb56e1679fc0bca \
|
||||
--hash=sha256:4647afc2f90d1ddd33441e5b0e85b16b12ddec4fca55f0d9671fef036ecca27c \
|
||||
--hash=sha256:4671d9dd5ec934cb9a73e7ee9676f9362aba54f7f34910956b84d727b0d73fb6 \
|
||||
--hash=sha256:53f77cbe57044e88bbd5ed26ac1d0514d2acf0591dd6bb02a3ae37f76811b80c \
|
||||
--hash=sha256:5eda85d6d1879e692d546a078b44251cdd08dd1cfb98dfb77b670c97cee49ea0 \
|
||||
--hash=sha256:5fed36fccc0612a53f1d4d9a816b50a36702c28a2aa880cb8a122b3466638743 \
|
||||
--hash=sha256:61d028e90346df14fedc3d1e5441df818d095f3b87d286825dfcbd6459b7ef63 \
|
||||
--hash=sha256:66f011380d0e49ed280c789fbd08ff0d40968ee7b665575489afa95c98196ab5 \
|
||||
--hash=sha256:6824f87845e3396029f3820c206e459ccc91760e8fa24422f8b0c3d1731cbec5 \
|
||||
--hash=sha256:6c6c373cfc5c83a975506110d17457138c8c63016b563cc9ed6e056a82f13ce4 \
|
||||
--hash=sha256:6d02d6655b0e54f54c4ef0b94eb6be0607b70853c45ce98bd278dc7de718be5d \
|
||||
--hash=sha256:6d50360be4546678fc1b79ffe7a66265e28667840010348dd69a314145807a1b \
|
||||
--hash=sha256:730cacb21e1bdff3ce90babf007d0a0917cc3e6492f336c2f0134101e0944f93 \
|
||||
--hash=sha256:737fe7d37e1a1bffe70bd5754ea763a62a066dc5913ca57e957824b72a85e205 \
|
||||
--hash=sha256:74a03b9698e198d47562765773b4a8309919089150a0bb17d829ad7b44b60d27 \
|
||||
--hash=sha256:7553fb2090d71822f02c629afe6042c299edf91ba1bf94951165613553984512 \
|
||||
--hash=sha256:7a66c7204d8869299919db4d5069a82f1561581af12b11b3c9f48c584eb8743d \
|
||||
--hash=sha256:7cc09976e8b56f8cebd752f7113ad07752461f48a58cbba644139015ac24954c \
|
||||
--hash=sha256:81afed14892743bbe14dacb9e36d9e0e504cd204e0b165062c488942b9718037 \
|
||||
--hash=sha256:8941aaadaf67246224cee8c3803777eed332a19d909b47e29c9842ef1e79ac26 \
|
||||
--hash=sha256:89472c9762729b5ae1ad974b777416bfda4ac5642423fa93bd57a09204712322 \
|
||||
--hash=sha256:8ea985900c5c95ce9db1745f7933eeef5d314f0565b27625d9a10ec9881e1bfb \
|
||||
--hash=sha256:8eca2a813c1cb7ad4fb74d368c2ffbbb4789d377ee5bb8df98373c2cc0dee76c \
|
||||
--hash=sha256:92b68146a71df78564e4ef48af17551a5ddd142e5190cdf2c5624d0c3ff5b2e8 \
|
||||
--hash=sha256:9332088d75dc3241c702d852d4671613136d90fa6881da7d770a483fd05248b4 \
|
||||
--hash=sha256:94698a9c5f91f9d138526b48fe26a199609544591f859c870d477351dc7b2414 \
|
||||
--hash=sha256:9a67fc9e8eb39039280526379fb3a70023d77caec1852002b4da7e8b270c4dd9 \
|
||||
--hash=sha256:9de40a7b0323d889cf8d23d1ef214f565ab154443c42737dfe52ff82cf857664 \
|
||||
--hash=sha256:a05d0c237b3349096d3981b727493e22147f934b20f6f125a3eba8f994bec4a9 \
|
||||
--hash=sha256:afb8db5439b81cf9c9d0c80404b60c3cc9c3add93e114dcae767f1477cb53775 \
|
||||
--hash=sha256:b18a3ed7d5b3bd8d9ef7a8cb226502c6bf8308df1525e1cc676c3680e7176739 \
|
||||
--hash=sha256:b1e74d11748e7e98e2f426ab176d4ed720a64412b6a15054378afdb71e0f37dc \
|
||||
--hash=sha256:b21e08af67b8a103c71a250401c78d5e0893beff75e28c53c98f4de42f774062 \
|
||||
--hash=sha256:b4c854ef3adc177950a8dfc81a86f5115d2abd545751a304c5bcf2c2c7283cfe \
|
||||
--hash=sha256:b882b3df248017dba09d6b16defe9b5c407fe32fc7c65a9c69798e6175601be9 \
|
||||
--hash=sha256:baf5215e0ab74c16e2dd324e8ec067ef59e41125d3eade2b863d294fd5035c92 \
|
||||
--hash=sha256:c649e3a33450ec82378822b3dad03cc228b8f5963c0c12fc3b1e0ab940f768a5 \
|
||||
--hash=sha256:c654de545946e0db659b3400168c9ad31b5d29593291482c43e3564effbcee13 \
|
||||
--hash=sha256:c6638687455baf640e37344fe26d37c404db8b80d037c3d29f58fe8d1c3b194d \
|
||||
--hash=sha256:c8d3b5532fc71b7a77c09192b4a5a200ea992702734a2e9279a37f2478236f26 \
|
||||
--hash=sha256:cb527a79772e5ef98fb1d700678fe031e353e765d1ca2d409c92263c6d43e09f \
|
||||
--hash=sha256:cf364028c016c03078a23b503f02058f1814320a56ad535686f90565636a9495 \
|
||||
--hash=sha256:d48a880098c96020b02d5a1f7d9251308510ce8858940e6fa99ece33f610838b \
|
||||
--hash=sha256:d68b6cef7827e8641e8ef16f4494edda8b36104d79773a334beaa1e3521430f6 \
|
||||
--hash=sha256:d9b29c1f0ae438d5ee9acb31cadee00a58c46cc9c0b2f9038c6b0b3470877a8c \
|
||||
--hash=sha256:d9b97165e8aed9272a6bb17c01e3cc5871a594a446ebedc996e2397a1c1ea8ef \
|
||||
--hash=sha256:da68248800ad6320861f129cd9c1bf96ca849a2771a59e0344e88681905916f5 \
|
||||
--hash=sha256:da902562c3e9c550df360bfa53c035b2f241fed6d9aef119048073680ace4a18 \
|
||||
--hash=sha256:dbd5c7a25a7cb98f5ca55d258b103a2054f859a46ae11aaf23134f9cc0d356ad \
|
||||
--hash=sha256:dd4f05f54a52fb558f1ba9f528228066954fee3ebe629fc1660d874d040ae5a3 \
|
||||
--hash=sha256:de8dad4425a6ca6e4e5e297b27b5c824ecc7581910bf9aee86cb6835e6812aa7 \
|
||||
--hash=sha256:e11e82b744887154b182fd3e7e8512418446501191994dbf9c9fc1f32cc8efd5 \
|
||||
--hash=sha256:e6e73b9e02893c764e7e8d5bb5ce277f1a009cd5243f8228f75f842bf937c534 \
|
||||
--hash=sha256:f73b96c41e3b2adedc34a7356e64c8eb96e03a3782b535e043a986276ce12a49 \
|
||||
--hash=sha256:f93fd8e5c8c0a4aa1f424d6173f14a892044054871c771f8566e4008eaa359d2 \
|
||||
--hash=sha256:fc33c5141b55ed366cfaad382df24fe7dcbc686de5be719b207bb248e3053dc5 \
|
||||
--hash=sha256:fc7de24befaeae77ba923797c7c87834c73648a05a4bde34b3b7e5588973a453 \
|
||||
--hash=sha256:fe562eb1a64e67dd297ccc4f5addea2501664954f2692b69a76449ec7913ecbf
|
||||
cffi==1.15.1 \
|
||||
--hash=sha256:00a9ed42e88df81ffae7a8ab6d9356b371399b91dbdf0c3cb1e84c03a13aceb5 \
|
||||
--hash=sha256:03425bdae262c76aad70202debd780501fabeaca237cdfddc008987c0e0f59ef \
|
||||
--hash=sha256:04ed324bda3cda42b9b695d51bb7d54b680b9719cfab04227cdd1e04e5de3104 \
|
||||
--hash=sha256:0e2642fe3142e4cc4af0799748233ad6da94c62a8bec3a6648bf8ee68b1c7426 \
|
||||
--hash=sha256:173379135477dc8cac4bc58f45db08ab45d228b3363adb7af79436135d028405 \
|
||||
--hash=sha256:198caafb44239b60e252492445da556afafc7d1e3ab7a1fb3f0584ef6d742375 \
|
||||
--hash=sha256:1e74c6b51a9ed6589199c787bf5f9875612ca4a8a0785fb2d4a84429badaf22a \
|
||||
--hash=sha256:2012c72d854c2d03e45d06ae57f40d78e5770d252f195b93f581acf3ba44496e \
|
||||
--hash=sha256:21157295583fe8943475029ed5abdcf71eb3911894724e360acff1d61c1d54bc \
|
||||
--hash=sha256:2470043b93ff09bf8fb1d46d1cb756ce6132c54826661a32d4e4d132e1977adf \
|
||||
--hash=sha256:285d29981935eb726a4399badae8f0ffdff4f5050eaa6d0cfc3f64b857b77185 \
|
||||
--hash=sha256:30d78fbc8ebf9c92c9b7823ee18eb92f2e6ef79b45ac84db507f52fbe3ec4497 \
|
||||
--hash=sha256:320dab6e7cb2eacdf0e658569d2575c4dad258c0fcc794f46215e1e39f90f2c3 \
|
||||
--hash=sha256:33ab79603146aace82c2427da5ca6e58f2b3f2fb5da893ceac0c42218a40be35 \
|
||||
--hash=sha256:3548db281cd7d2561c9ad9984681c95f7b0e38881201e157833a2342c30d5e8c \
|
||||
--hash=sha256:3799aecf2e17cf585d977b780ce79ff0dc9b78d799fc694221ce814c2c19db83 \
|
||||
--hash=sha256:39d39875251ca8f612b6f33e6b1195af86d1b3e60086068be9cc053aa4376e21 \
|
||||
--hash=sha256:3b926aa83d1edb5aa5b427b4053dc420ec295a08e40911296b9eb1b6170f6cca \
|
||||
--hash=sha256:3bcde07039e586f91b45c88f8583ea7cf7a0770df3a1649627bf598332cb6984 \
|
||||
--hash=sha256:3d08afd128ddaa624a48cf2b859afef385b720bb4b43df214f85616922e6a5ac \
|
||||
--hash=sha256:3eb6971dcff08619f8d91607cfc726518b6fa2a9eba42856be181c6d0d9515fd \
|
||||
--hash=sha256:40f4774f5a9d4f5e344f31a32b5096977b5d48560c5592e2f3d2c4374bd543ee \
|
||||
--hash=sha256:4289fc34b2f5316fbb762d75362931e351941fa95fa18789191b33fc4cf9504a \
|
||||
--hash=sha256:470c103ae716238bbe698d67ad020e1db9d9dba34fa5a899b5e21577e6d52ed2 \
|
||||
--hash=sha256:4f2c9f67e9821cad2e5f480bc8d83b8742896f1242dba247911072d4fa94c192 \
|
||||
--hash=sha256:50a74364d85fd319352182ef59c5c790484a336f6db772c1a9231f1c3ed0cbd7 \
|
||||
--hash=sha256:54a2db7b78338edd780e7ef7f9f6c442500fb0d41a5a4ea24fff1c929d5af585 \
|
||||
--hash=sha256:5635bd9cb9731e6d4a1132a498dd34f764034a8ce60cef4f5319c0541159392f \
|
||||
--hash=sha256:59c0b02d0a6c384d453fece7566d1c7e6b7bae4fc5874ef2ef46d56776d61c9e \
|
||||
--hash=sha256:5d598b938678ebf3c67377cdd45e09d431369c3b1a5b331058c338e201f12b27 \
|
||||
--hash=sha256:5df2768244d19ab7f60546d0c7c63ce1581f7af8b5de3eb3004b9b6fc8a9f84b \
|
||||
--hash=sha256:5ef34d190326c3b1f822a5b7a45f6c4535e2f47ed06fec77d3d799c450b2651e \
|
||||
--hash=sha256:6975a3fac6bc83c4a65c9f9fcab9e47019a11d3d2cf7f3c0d03431bf145a941e \
|
||||
--hash=sha256:6c9a799e985904922a4d207a94eae35c78ebae90e128f0c4e521ce339396be9d \
|
||||
--hash=sha256:70df4e3b545a17496c9b3f41f5115e69a4f2e77e94e1d2a8e1070bc0c38c8a3c \
|
||||
--hash=sha256:7473e861101c9e72452f9bf8acb984947aa1661a7704553a9f6e4baa5ba64415 \
|
||||
--hash=sha256:8102eaf27e1e448db915d08afa8b41d6c7ca7a04b7d73af6514df10a3e74bd82 \
|
||||
--hash=sha256:87c450779d0914f2861b8526e035c5e6da0a3199d8f1add1a665e1cbc6fc6d02 \
|
||||
--hash=sha256:8b7ee99e510d7b66cdb6c593f21c043c248537a32e0bedf02e01e9553a172314 \
|
||||
--hash=sha256:91fc98adde3d7881af9b59ed0294046f3806221863722ba7d8d120c575314325 \
|
||||
--hash=sha256:94411f22c3985acaec6f83c6df553f2dbe17b698cc7f8ae751ff2237d96b9e3c \
|
||||
--hash=sha256:98d85c6a2bef81588d9227dde12db8a7f47f639f4a17c9ae08e773aa9c697bf3 \
|
||||
--hash=sha256:9ad5db27f9cabae298d151c85cf2bad1d359a1b9c686a275df03385758e2f914 \
|
||||
--hash=sha256:a0b71b1b8fbf2b96e41c4d990244165e2c9be83d54962a9a1d118fd8657d2045 \
|
||||
--hash=sha256:a0f100c8912c114ff53e1202d0078b425bee3649ae34d7b070e9697f93c5d52d \
|
||||
--hash=sha256:a591fe9e525846e4d154205572a029f653ada1a78b93697f3b5a8f1f2bc055b9 \
|
||||
--hash=sha256:a5c84c68147988265e60416b57fc83425a78058853509c1b0629c180094904a5 \
|
||||
--hash=sha256:a66d3508133af6e8548451b25058d5812812ec3798c886bf38ed24a98216fab2 \
|
||||
--hash=sha256:a8c4917bd7ad33e8eb21e9a5bbba979b49d9a97acb3a803092cbc1133e20343c \
|
||||
--hash=sha256:b3bbeb01c2b273cca1e1e0c5df57f12dce9a4dd331b4fa1635b8bec26350bde3 \
|
||||
--hash=sha256:cba9d6b9a7d64d4bd46167096fc9d2f835e25d7e4c121fb2ddfc6528fb0413b2 \
|
||||
--hash=sha256:cc4d65aeeaa04136a12677d3dd0b1c0c94dc43abac5860ab33cceb42b801c1e8 \
|
||||
--hash=sha256:ce4bcc037df4fc5e3d184794f27bdaab018943698f4ca31630bc7f84a7b69c6d \
|
||||
--hash=sha256:cec7d9412a9102bdc577382c3929b337320c4c4c4849f2c5cdd14d7368c5562d \
|
||||
--hash=sha256:d400bfb9a37b1351253cb402671cea7e89bdecc294e8016a707f6d1d8ac934f9 \
|
||||
--hash=sha256:d61f4695e6c866a23a21acab0509af1cdfd2c013cf256bbf5b6b5e2695827162 \
|
||||
--hash=sha256:db0fbb9c62743ce59a9ff687eb5f4afbe77e5e8403d6697f7446e5f609976f76 \
|
||||
--hash=sha256:dd86c085fae2efd48ac91dd7ccffcfc0571387fe1193d33b6394db7ef31fe2a4 \
|
||||
--hash=sha256:e00b098126fd45523dd056d2efba6c5a63b71ffe9f2bbe1a4fe1716e1d0c331e \
|
||||
--hash=sha256:e229a521186c75c8ad9490854fd8bbdd9a0c9aa3a524326b55be83b54d4e0ad9 \
|
||||
--hash=sha256:e263d77ee3dd201c3a142934a086a4450861778baaeeb45db4591ef65550b0a6 \
|
||||
--hash=sha256:ed9cb427ba5504c1dc15ede7d516b84757c3e3d7868ccc85121d9310d27eed0b \
|
||||
--hash=sha256:fa6693661a4c91757f4412306191b6dc88c1703f780c8234035eac011922bc01 \
|
||||
--hash=sha256:fcd131dd944808b5bdb38e6f5b53013c5aa4f334c5cad0c72742f6eba4b73db0
|
||||
# via
|
||||
# cryptography
|
||||
# pynacl
|
||||
@@ -210,53 +185,44 @@ click==8.1.7 \
|
||||
--hash=sha256:ae74fb96c20a0277a1d615f1e4d73c8414f5a98db8b799a7931d1582f3390c28 \
|
||||
--hash=sha256:ca9853ad459e787e2192211578cc907e7594e294c7ccc834310722b41b9ca6de
|
||||
# via black
|
||||
cryptography==50.0.0 \
|
||||
--hash=sha256:031e2d5dd4bb9caa3ca9c82e5a197fd8ae680232cee62603d1a813f3f07e3d03 \
|
||||
--hash=sha256:06a32a980526a6ab9a4b9bf8f7385800791e2bb960903cb6b530e4817509a3b7 \
|
||||
--hash=sha256:07479a1cb08219ab719147e742e76090c9c773321959bb94946fffdd397a6437 \
|
||||
--hash=sha256:07949c449a1abcf60d1ee6e88956d89404c7df3c8258f46589e912988e551987 \
|
||||
--hash=sha256:105110f43a471dbd0060b9c9516cb8a6a79233631a04cc2ba16f28323ac6e025 \
|
||||
--hash=sha256:11b74db56cdbe3cdee6e3f6982ecb70334fa10dce99ed58bf7894aaaa3b2a037 \
|
||||
--hash=sha256:12b9c6996425c76ea6c457ace4f3073e715b8c545add07cd1a8f3a4f90691269 \
|
||||
--hash=sha256:1489e263a8048bb8b6a8bac662eb2d402ea5d2b7b4699b72f385f1e2772db105 \
|
||||
--hash=sha256:19736989797678c6af1e55cd49055cdbcb55d8f6b5583ac5335f933aba9101dc \
|
||||
--hash=sha256:1b4a266766514614f8aa60416e71f2fc6e575d36e7bdc90f644fadb2f4b75b95 \
|
||||
--hash=sha256:2a8183b489dc1f7f80f135780fadc1108f14b31b8a40411c7a5b17425f65f28b \
|
||||
--hash=sha256:37fdb0d0111f1e2ff07139dfb79f1b49531f8e213c46f1163dd7642979b58c47 \
|
||||
--hash=sha256:3f5735ffe4996d28b809371756219f5354864902a3b9e7c0b9ee87041209fc9c \
|
||||
--hash=sha256:49e7d93abdbd2990caced757e5fade25302f719c3c8fb6e6fff2dde98999fc41 \
|
||||
--hash=sha256:5e34edd123674534acd70147f0ca331eaa2c74e6325fb2028c886aa26ba0b68c \
|
||||
--hash=sha256:62598a8a57f815db4c6259a4e97d857dab56697e7de8e8ab02352ab74da1995d \
|
||||
--hash=sha256:65c2c3add92b45fd0709db8594536aea39c2a67af0e27ffcf049c498501140b7 \
|
||||
--hash=sha256:6ba6a53445bd3cfa809ef3ef5f1589aa6ba08784a1d962bf47d0940e871dab1c \
|
||||
--hash=sha256:6e7d61120573a7f2cd94cc095f9e81f6967c61ccdf194285aa143ecec8e0b708 \
|
||||
--hash=sha256:7cec5b856506da6defb290f30c9ee687d5f5e8cb0bd3f6459dde43b0b4fa40ef \
|
||||
--hash=sha256:80b63928fa35083b33966ce1efb70e5b9607181e49dcd1c22c8c005e319f667f \
|
||||
--hash=sha256:82148ec5bddac30b51a5b3c1945075f896fa022cb93f8e4a01e9f6ee95292c5f \
|
||||
--hash=sha256:828743d939e9629bc267b8e2d08d8bb67cd4319c771a33d4b18b22dd8fb7440a \
|
||||
--hash=sha256:8d89f3976b10b4ce31118de72329025f70d2c6ead14a8217c5514dd2c6d5a78f \
|
||||
--hash=sha256:8eb5e1172eb569ea8a872796576e6a67c276351728b6455d5beb01242b027c6a \
|
||||
--hash=sha256:900131fafd8aead39ac7dd3a7e833be754c17a95cfd91221636949fe4eb0aa8a \
|
||||
--hash=sha256:910d11e1a385c654bf738bf3e6b8e6ed5de0f5610fcae2be9e5b398d8081d20e \
|
||||
--hash=sha256:910e1d2668e7de9648f2bcee30e180db2a6b15c30f887d7c4c93ddf96e3992e3 \
|
||||
--hash=sha256:9aa87839c383bdbab6ef865787a1fb877af8dd03464c4400322726feaaadfc6d \
|
||||
--hash=sha256:a1b30560f2acc95aa8b2e06e716a13dbfc97314747b80d9707e307f77b40d6b3 \
|
||||
--hash=sha256:a91296cb61e8df6f86d0c19cc4068228da256bf59bf86049fbd821084565327f \
|
||||
--hash=sha256:b42a28c1844fd9de8f3f7d540e36b66f3a9c83fceac7170ebc7a6a19edd9dcae \
|
||||
--hash=sha256:bd1c592e4d5974f0d08d4888e432157adba757c66da0246918e43677fafa2d30 \
|
||||
--hash=sha256:c87f62a3d3b9888ed0fdde100ec06aa61ca9cd44bad9057d1dff9a516b5f5bb9 \
|
||||
--hash=sha256:c99c003e088647b8a5b7c145d6f78c335f6348332b62e142d411c4b63d1460b9 \
|
||||
--hash=sha256:ccdc4a71a4dabae05de219404f9f4abc38e3b58422177ff93d0da05967dafa07 \
|
||||
--hash=sha256:d24fead1d4d076e1bfb006dcec392074a3cd8d7b4fc8a595aa64073b2b7a96ba \
|
||||
--hash=sha256:d58c3db7cd6eed54e6c06744db55456b65ebd7492ddeae9c1e93cfca7aa857d3 \
|
||||
--hash=sha256:d764dcf130c428ef66786f866dd750f53182bc608813489915e9fc106bb0c82f \
|
||||
--hash=sha256:df2a58a472f332225671c35b0a830208b86d004f82baa8530fa3782c85646533 \
|
||||
--hash=sha256:e722f16708d854fe924790e051061f6704a472c3bac347b6fd88033ea8dd0dc5 \
|
||||
--hash=sha256:ecfed7367f965a0328cfbdd70da860f15441f002f613185668c6e6ebf5a0ac11 \
|
||||
--hash=sha256:eeac2acb5a20ed25e0ad6d1df9891a520b78b404266b6d11778f25d5d691a6c9 \
|
||||
--hash=sha256:f59e38625469987d7ef6d495323c55e7db6c212eaf6112267e0d3b565a2e9c9f \
|
||||
--hash=sha256:f89831ef99dd7dd169ab06d63a831adb9e20a87aac6d380266bbda5823349169 \
|
||||
--hash=sha256:fd9192b7b70c573d7f214eb1ae35e00d359f6f5e4b27c7e21e30de1fc6204645
|
||||
cryptography==45.0.5 \
|
||||
--hash=sha256:0027d566d65a38497bc37e0dd7c2f8ceda73597d2ac9ba93810204f56f52ebc7 \
|
||||
--hash=sha256:101ee65078f6dd3e5a028d4f19c07ffa4dd22cce6a20eaa160f8b5219911e7d8 \
|
||||
--hash=sha256:12e55281d993a793b0e883066f590c1ae1e802e3acb67f8b442e721e475e6463 \
|
||||
--hash=sha256:14d96584701a887763384f3c47f0ca7c1cce322aa1c31172680eb596b890ec30 \
|
||||
--hash=sha256:1e1da5accc0c750056c556a93c3e9cb828970206c68867712ca5805e46dc806f \
|
||||
--hash=sha256:206210d03c1193f4e1ff681d22885181d47efa1ab3018766a7b32a7b3d6e6afd \
|
||||
--hash=sha256:2089cc8f70a6e454601525e5bf2779e665d7865af002a5dec8d14e561002e135 \
|
||||
--hash=sha256:3a264aae5f7fbb089dbc01e0242d3b67dffe3e6292e1f5182122bdf58e65215d \
|
||||
--hash=sha256:3af26738f2db354aafe492fb3869e955b12b2ef2e16908c8b9cb928128d42c57 \
|
||||
--hash=sha256:3fcfbefc4a7f332dece7272a88e410f611e79458fab97b5efe14e54fe476f4fd \
|
||||
--hash=sha256:460f8c39ba66af7db0545a8c6f2eabcbc5a5528fc1cf6c3fa9a1e44cec33385e \
|
||||
--hash=sha256:57c816dfbd1659a367831baca4b775b2a5b43c003daf52e9d57e1d30bc2e1b0e \
|
||||
--hash=sha256:5aa1e32983d4443e310f726ee4b071ab7569f58eedfdd65e9675484a4eb67bd1 \
|
||||
--hash=sha256:6ff8728d8d890b3dda5765276d1bc6fb099252915a2cd3aff960c4c195745dd0 \
|
||||
--hash=sha256:7259038202a47fdecee7e62e0fd0b0738b6daa335354396c6ddebdbe1206af2a \
|
||||
--hash=sha256:72e76caa004ab63accdf26023fccd1d087f6d90ec6048ff33ad0445abf7f605a \
|
||||
--hash=sha256:7760c1c2e1a7084153a0f68fab76e754083b126a47d0117c9ed15e69e2103492 \
|
||||
--hash=sha256:8c4a6ff8a30e9e3d38ac0539e9a9e02540ab3f827a3394f8852432f6b0ea152e \
|
||||
--hash=sha256:9024beb59aca9d31d36fcdc1604dd9bbeed0a55bface9f1908df19178e2f116e \
|
||||
--hash=sha256:90cb0a7bb35959f37e23303b7eed0a32280510030daba3f7fdfbb65defde6a97 \
|
||||
--hash=sha256:91098f02ca81579c85f66df8a588c78f331ca19089763d733e34ad359f474174 \
|
||||
--hash=sha256:926c3ea71a6043921050eaa639137e13dbe7b4ab25800932a8498364fc1abec9 \
|
||||
--hash=sha256:982518cd64c54fcada9d7e5cf28eabd3ee76bd03ab18e08a48cad7e8b6f31b18 \
|
||||
--hash=sha256:9b4cf6318915dccfe218e69bbec417fdd7c7185aa7aab139a2c0beb7468c89f0 \
|
||||
--hash=sha256:ad0caded895a00261a5b4aa9af828baede54638754b51955a0ac75576b831b27 \
|
||||
--hash=sha256:b85980d1e345fe769cfc57c57db2b59cff5464ee0c045d52c0df087e926fbe63 \
|
||||
--hash=sha256:b8fa8b0a35a9982a3c60ec79905ba5bb090fc0b9addcfd3dc2dd04267e45f25e \
|
||||
--hash=sha256:b9e38e0a83cd51e07f5a48ff9691cae95a79bea28fe4ded168a8e5c6c77e819d \
|
||||
--hash=sha256:bd4c45986472694e5121084c6ebbd112aa919a25e783b87eb95953c9573906d6 \
|
||||
--hash=sha256:be97d3a19c16a9be00edf79dca949c8fa7eff621763666a145f9f9535a5d7f42 \
|
||||
--hash=sha256:c648025b6840fe62e57107e0a25f604db740e728bd67da4f6f060f03017d5097 \
|
||||
--hash=sha256:d05a38884db2ba215218745f0781775806bde4f32e07b135348355fe8e4991d9 \
|
||||
--hash=sha256:dd420e577921c8c2d31289536c386aaa30140b473835e97f83bc71ea9d2baf2d \
|
||||
--hash=sha256:e357286c1b76403dd384d938f93c46b2b058ed4dfcdce64a770f0537ed3feb6f \
|
||||
--hash=sha256:e6c00130ed423201c5bc5544c23359141660b07999ad82e34e7bb8f882bb78e0 \
|
||||
--hash=sha256:e74d30ec9c7cb2f404af331d5b4099a9b322a8a6b25c4632755c8757345baac5 \
|
||||
--hash=sha256:f3562c2f23c612f2e4a6964a61d942f891d29ee320edb62ff48ffb99f3de9ae8
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pyjwt
|
||||
@@ -278,9 +244,9 @@ graylint==1.1.1 \
|
||||
--hash=sha256:0fd8e02972ca03d0ef2bf0adea76b5343efcd492d7afb5f658f3e3a724f55a36 \
|
||||
--hash=sha256:b7e0eab6c159684dbf5ef84e942c3340f6a6549b02a3d11b1a1763cc4f8f0593
|
||||
# via darker
|
||||
idna==3.16 \
|
||||
--hash=sha256:cc246e3a3f89580c3a951b5ad298ca4638078b2cdd4f115654332b5c26daded5 \
|
||||
--hash=sha256:d7a6da03db833450fca25d2358ac9ff06cd624577a4aea3a596d5c0f77b8e03d
|
||||
idna==3.10 \
|
||||
--hash=sha256:12f65c9b470abda6dc35cf8e63cc574b1c52b11df2c86030af0ac09b01b13ea9 \
|
||||
--hash=sha256:946d195a0d259cbba61165e88e65941f16e9b36ea6ddb97f00452bae8b1287d3
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# requests
|
||||
@@ -292,9 +258,9 @@ packaging==23.1 \
|
||||
--hash=sha256:994793af429502c4ea2ebf6bf664629d07c1a9fe974af92966e4b8d2df7edc61 \
|
||||
--hash=sha256:a392980d2b6cffa644431898be54b0045151319d1e7ec34f0cfed48767dd334f
|
||||
# via black
|
||||
pathspec==1.0.4 \
|
||||
--hash=sha256:0210e2ae8a21a9137c0d470578cb0e595af87edaa6ebf12ff176f14a02e0e645 \
|
||||
--hash=sha256:fb6ae2fd4e7c921a165808a552060e722767cfa526f99ca5156ed2ce45a5c723
|
||||
pathspec==0.11.2 \
|
||||
--hash=sha256:1d6ed233af05e679efb96b1851550ea95bbb64b7c490b0f5aa52996c11e92a20 \
|
||||
--hash=sha256:e0d8d0ac2f12da61956eb2306b69f9469b42f4deb0f3cb6ed47b9cce9996ced3
|
||||
# via black
|
||||
platformdirs==3.10.0 \
|
||||
--hash=sha256:b45696dab2d7cc691a3226759c0d3b00c47c8b6e293d96f6436f733303f77f6d \
|
||||
@@ -308,88 +274,25 @@ pygithub==2.6.1 \
|
||||
--hash=sha256:6f2fa6d076ccae475f9fc392cc6cdbd54db985d4f69b8833a28397de75ed6ca3 \
|
||||
--hash=sha256:b5c035392991cca63959e9453286b41b54d83bf2de2daa7d7ff7e4312cebf3bf
|
||||
# via -r requirements_formatting.txt.in
|
||||
pyjwt==2.13.0 \
|
||||
--hash=sha256:41571c89ca91598c79e8ef18a2d07367d4810fbbd6f637794879baf1b7703423 \
|
||||
--hash=sha256:66adcc2aff09b3f1bbd95fc1e1577df8ac8723c978552fd43304c8a290ac5728
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
pynacl==1.6.2 \
|
||||
--hash=sha256:018494d6d696ae03c7e656e5e74cdfd8ea1326962cc401bcf018f1ed8436811c \
|
||||
--hash=sha256:04316d1fc625d860b6c162fff704eb8426b1a8bcd3abacea11142cbd99a6b574 \
|
||||
--hash=sha256:22de65bb9010a725b0dac248f353bb072969c94fa8d6b1f34b87d7953cf7bbe4 \
|
||||
--hash=sha256:26bfcd00dcf2cf160f122186af731ae30ab120c18e8375684ec2670dccd28130 \
|
||||
--hash=sha256:2fef529ef3ee487ad8113d287a593fa26f48ee3620d92ecc6f1d09ea38e0709b \
|
||||
--hash=sha256:320ef68a41c87547c91a8b58903c9caa641ab01e8512ce291085b5fe2fcb7590 \
|
||||
--hash=sha256:3bffb6d0f6becacb6526f8f42adfb5efb26337056ee0831fb9a7044d1a964444 \
|
||||
--hash=sha256:44081faff368d6c5553ccf55322ef2819abb40e25afaec7e740f159f74813634 \
|
||||
--hash=sha256:46065496ab748469cdd999246d17e301b2c24ae2fdf739132e580a0e94c94a87 \
|
||||
--hash=sha256:5811c72b473b2f38f7e2a3dc4f8642e3a3e9b5e7317266e4ced1fba85cae41aa \
|
||||
--hash=sha256:622d7b07cc5c02c666795792931b50c91f3ce3c2649762efb1ef0d5684c81594 \
|
||||
--hash=sha256:62985f233210dee6548c223301b6c25440852e13d59a8b81490203c3227c5ba0 \
|
||||
--hash=sha256:68be3a09455743ff9505491220b64440ced8973fe930f270c8e07ccfa25b1f9e \
|
||||
--hash=sha256:834a43af110f743a754448463e8fd61259cd4ab5bbedcf70f9dabad1d28a394c \
|
||||
--hash=sha256:8845c0631c0be43abdd865511c41eab235e0be69c81dc66a50911594198679b0 \
|
||||
--hash=sha256:8a66d6fb6ae7661c58995f9c6435bda2b1e68b54b598a6a10247bfcdadac996c \
|
||||
--hash=sha256:8b097553b380236d51ed11356c953bf8ce36a29a3e596e934ecabe76c985a577 \
|
||||
--hash=sha256:a84bf1c20339d06dc0c85d9aea9637a24f718f375d861b2668b2f9f96fa51145 \
|
||||
--hash=sha256:a9f9932d8d2811ce1a8ffa79dcbdf3970e7355b5c8eb0c1a881a57e7f7d96e88 \
|
||||
--hash=sha256:bc4a36b28dd72fb4845e5d8f9760610588a96d5a51f01d84d8c6ff9849968c14 \
|
||||
--hash=sha256:c8a231e36ec2cab018c4ad4358c386e36eede0319a0c41fed24f840b1dac59f6 \
|
||||
--hash=sha256:c949ea47e4206af7c8f604b8278093b674f7c79ed0d4719cc836902bf4517465 \
|
||||
--hash=sha256:d071c6a9a4c94d79eb665db4ce5cedc537faf74f2355e4d502591d850d3913c0 \
|
||||
--hash=sha256:d29bfe37e20e015a7d8b23cfc8bd6aa7909c92a1b8f41ee416bbb3e79ef182b2 \
|
||||
--hash=sha256:fe9847ca47d287af41e82be1dd5e23023d3c31a951da134121ab02e42ac218c9
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
pytokens==0.4.1 \
|
||||
--hash=sha256:0fc71786e629cef478cbf29d7ea1923299181d0699dbe7c3c0f4a583811d9fc1 \
|
||||
--hash=sha256:11edda0942da80ff58c4408407616a310adecae1ddd22eef8c692fe266fa5009 \
|
||||
--hash=sha256:140709331e846b728475786df8aeb27d24f48cbcf7bcd449f8de75cae7a45083 \
|
||||
--hash=sha256:24afde1f53d95348b5a0eb19488661147285ca4dd7ed752bbc3e1c6242a304d1 \
|
||||
--hash=sha256:26cef14744a8385f35d0e095dc8b3a7583f6c953c2e3d269c7f82484bf5ad2de \
|
||||
--hash=sha256:27b83ad28825978742beef057bfe406ad6ed524b2d28c252c5de7b4a6dd48fa2 \
|
||||
--hash=sha256:292052fe80923aae2260c073f822ceba21f3872ced9a68bb7953b348e561179a \
|
||||
--hash=sha256:29d1d8fb1030af4d231789959f21821ab6325e463f0503a61d204343c9b355d1 \
|
||||
--hash=sha256:2a44ed93ea23415c54f3face3b65ef2b844d96aeb3455b8a69b3df6beab6acc5 \
|
||||
--hash=sha256:30f51edd9bb7f85c748979384165601d028b84f7bd13fe14d3e065304093916a \
|
||||
--hash=sha256:34bcc734bd2f2d5fe3b34e7b3c0116bfb2397f2d9666139988e7a3eb5f7400e3 \
|
||||
--hash=sha256:3ad72b851e781478366288743198101e5eb34a414f1d5627cdd585ca3b25f1db \
|
||||
--hash=sha256:3f901fe783e06e48e8cbdc82d631fca8f118333798193e026a50ce1b3757ea68 \
|
||||
--hash=sha256:42f144f3aafa5d92bad964d471a581651e28b24434d184871bd02e3a0d956037 \
|
||||
--hash=sha256:4a14d5f5fc78ce85e426aa159489e2d5961acf0e47575e08f35584009178e321 \
|
||||
--hash=sha256:4a58d057208cb9075c144950d789511220b07636dd2e4708d5645d24de666bdc \
|
||||
--hash=sha256:4e691d7f5186bd2842c14813f79f8884bb03f5995f0575272009982c5ac6c0f7 \
|
||||
--hash=sha256:5502408cab1cb18e128570f8d598981c68a50d0cbd7c61312a90507cd3a1276f \
|
||||
--hash=sha256:584c80c24b078eec1e227079d56dc22ff755e0ba8654d8383b2c549107528918 \
|
||||
--hash=sha256:5ad948d085ed6c16413eb5fec6b3e02fa00dc29a2534f088d3302c47eb59adf9 \
|
||||
--hash=sha256:670d286910b531c7b7e3c0b453fd8156f250adb140146d234a82219459b9640c \
|
||||
--hash=sha256:682fa37ff4d8e95f7df6fe6fe6a431e8ed8e788023c6bcc0f0880a12eab80ad1 \
|
||||
--hash=sha256:6d6c4268598f762bc8e91f5dbf2ab2f61f7b95bdc07953b602db879b3c8c18e1 \
|
||||
--hash=sha256:79fc6b8699564e1f9b521582c35435f1bd32dd06822322ec44afdeba666d8cb3 \
|
||||
--hash=sha256:8bdb9d0ce90cbf99c525e75a2fa415144fd570a1ba987380190e8b786bc6ef9b \
|
||||
--hash=sha256:8fcb9ba3709ff77e77f1c7022ff11d13553f3c30299a9fe246a166903e9091eb \
|
||||
--hash=sha256:941d4343bf27b605e9213b26bfa1c4bf197c9c599a9627eb7305b0defcfe40c1 \
|
||||
--hash=sha256:967cf6e3fd4adf7de8fc73cd3043754ae79c36475c1c11d514fc72cf5490094a \
|
||||
--hash=sha256:970b08dd6b86058b6dc07efe9e98414f5102974716232d10f32ff39701e841c4 \
|
||||
--hash=sha256:97f50fd18543be72da51dd505e2ed20d2228c74e0464e4262e4899797803d7fa \
|
||||
--hash=sha256:9bd7d7f544d362576be74f9d5901a22f317efc20046efe2034dced238cbbfe78 \
|
||||
--hash=sha256:add8bf86b71a5d9fb5b89f023a80b791e04fba57960aa790cc6125f7f1d39dfe \
|
||||
--hash=sha256:b35d7e5ad269804f6697727702da3c517bb8a5228afa450ab0fa787732055fc9 \
|
||||
--hash=sha256:b49750419d300e2b5a3813cf229d4e5a4c728dae470bcc89867a9ad6f25a722d \
|
||||
--hash=sha256:d31b97b3de0f61571a124a00ffe9a81fb9939146c122c11060725bd5aea79975 \
|
||||
--hash=sha256:d70e77c55ae8380c91c0c18dea05951482e263982911fc7410b1ffd1dadd3440 \
|
||||
--hash=sha256:d9907d61f15bf7261d7e775bd5d7ee4d2930e04424bab1972591918497623a16 \
|
||||
--hash=sha256:da5baeaf7116dced9c6bb76dc31ba04a2dc3695f3d9f74741d7910122b456edc \
|
||||
--hash=sha256:dc74c035f9bfca0255c1af77ddd2d6ae8419012805453e4b0e7513e17904545d \
|
||||
--hash=sha256:dcafc12c30dbaf1e2af0490978352e0c4041a7cde31f4f81435c2a5e8b9cabb6 \
|
||||
--hash=sha256:ee44d0f85b803321710f9239f335aafe16553b39106384cef8e6de40cb4ef2f6 \
|
||||
--hash=sha256:f66a6bbe741bd431f6d741e617e0f39ec7257ca1f89089593479347cc4d13324
|
||||
# via black
|
||||
requests==2.34.2 \
|
||||
--hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 \
|
||||
--hash=sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed
|
||||
pyjwt==2.8.0 \
|
||||
--hash=sha256:57e28d156e3d5c10088e0c68abb90bfac3df82b40a71bd0daa20c65ccd5c23de \
|
||||
--hash=sha256:59127c392cc44c2da5bb3192169a91f429924e17aff6534d70fdc02ab3e04320
|
||||
# via pygithub
|
||||
pynacl==1.5.0 \
|
||||
--hash=sha256:06b8f6fa7f5de8d5d2f7573fe8c863c051225a27b61e6860fd047b1775807858 \
|
||||
--hash=sha256:0c84947a22519e013607c9be43706dd42513f9e6ae5d39d3613ca1e142fba44d \
|
||||
--hash=sha256:20f42270d27e1b6a29f54032090b972d97f0a1b0948cc52392041ef7831fee93 \
|
||||
--hash=sha256:401002a4aaa07c9414132aaed7f6836ff98f59277a234704ff66878c2ee4a0d1 \
|
||||
--hash=sha256:52cb72a79269189d4e0dc537556f4740f7f0a9ec41c1322598799b0bdad4ef92 \
|
||||
--hash=sha256:61f642bf2378713e2c2e1de73444a3778e5f0a38be6fee0fe532fe30060282ff \
|
||||
--hash=sha256:8ac7448f09ab85811607bdd21ec2464495ac8b7c66d146bf545b0f08fb9220ba \
|
||||
--hash=sha256:a36d4a9dda1f19ce6e03c9a784a2921a4b726b02e1c736600ca9c22029474394 \
|
||||
--hash=sha256:a422368fc821589c228f4c49438a368831cb5bbc0eab5ebe1d7fac9dded6567b \
|
||||
--hash=sha256:e46dae94e34b085175f8abb3b0aaa7da40767865ac82c928eeb9e57e1ea8a543
|
||||
# via pygithub
|
||||
requests==2.32.4 \
|
||||
--hash=sha256:27babd3cda2a6d50b30443204ee89830707d396671944c998b5975b031ac2b2c \
|
||||
--hash=sha256:27d0316682c8a29834d3264820024b62a36942083d52caf2f14c0591336d3422
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
@@ -403,9 +306,9 @@ typing-extensions==4.14.1 \
|
||||
--hash=sha256:38b39f4aeeab64884ce9f74c94263ef78f3c22467c8724005483154c26648d36 \
|
||||
--hash=sha256:d1e1e3b58374dc93031d6eda2420a48ea44a36c2b4766a4fdeb3710755731d76
|
||||
# via pygithub
|
||||
urllib3==2.7.0 \
|
||||
--hash=sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c \
|
||||
--hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897
|
||||
urllib3==2.5.0 \
|
||||
--hash=sha256:3fc47733c7e419d4bc3f6b3dc2b4f890bb743906a30d56ba4a5bfa4bbff92760 \
|
||||
--hash=sha256:e6b01673c0fa6a13e374b50871808eb3bf7046c4b125b216f6bf1cc604cff0dc
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
|
||||
@@ -1,10 +1,8 @@
|
||||
black>=26.3.1
|
||||
black~=25.1
|
||||
darker==2.1.1
|
||||
PyGithub==2.6.1
|
||||
cryptography>=50.0.0
|
||||
urllib3>=2.7.0
|
||||
requests>=2.33.0
|
||||
idna>=3.15
|
||||
cryptography>=43.0.1
|
||||
urllib3>=2.5.0
|
||||
requests>=2.32.4
|
||||
idna>=3.7
|
||||
certifi>=2024.7.4
|
||||
PyNaCl>=1.6.2
|
||||
PyJWT>=2.13.0
|
||||
Vendored
+1
-1
Submodule External/drm-headers updated: 3e49836995...0675d2f291.
Vendored
+1
-1
Submodule External/fmt updated: c07e2aa4b1...20c8fdad06.
+1
Submodule External/jemalloc added at ce24593018.
+1
Submodule External/robin-map added at d5683d9f18.
Vendored
-1
Submodule External/rpmalloc deleted from 09142d7264.
Vendored
-3
@@ -1,6 +1,3 @@
|
||||
set(NAME tiny-json)
|
||||
set(SRCS tiny-json.c)
|
||||
add_library(${NAME} STATIC ${SRCS})
|
||||
|
||||
target_include_directories(${NAME} PUBLIC ${CMAKE_CURRENT_LIST_DIR})
|
||||
add_library(${NAME}::${NAME} ALIAS ${NAME})
|
||||
Vendored
-1
Submodule External/unordered_dense deleted from 3234af2c03.
Vendored
+1
-1
Submodule External/vixl updated: 20bccdbe04...ed690c9eca.
Vendored
+1
-1
Submodule External/xxhash updated: e626a72bc2...bbb27a5efb.
Vendored
-1
Submodule External/zydis deleted from 9bfadd6a55.
+42
-9
@@ -1,16 +1,16 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
set(PROJECT_NAME FEXCore)
|
||||
set (PROJECT_NAME FEXCore)
|
||||
project(${PROJECT_NAME}
|
||||
VERSION 0.01
|
||||
LANGUAGES CXX)
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(ARCHITECTURE_x86_64 1)
|
||||
set(_M_X86_64 1)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(ARCHITECTURE_arm64 1)
|
||||
set(_M_ARM_64 1)
|
||||
endif()
|
||||
|
||||
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
|
||||
@@ -24,10 +24,45 @@ include(CheckCXXCompilerFlag)
|
||||
include(CheckIncludeFileCXX)
|
||||
include(CheckCXXSourceCompiles)
|
||||
|
||||
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
|
||||
# Useful to have for freestanding libFEXCore
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(External/vixl/src/)
|
||||
endif()
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
|
||||
configure_file(${CMAKE_CURRENT_SOURCE_DIR}/include/git_version.h.in
|
||||
set(GIT_SHORT_HASH "Unknown")
|
||||
set(GIT_DESCRIBE_STRING "FEX-Unknown")
|
||||
|
||||
if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
# Find our git hash
|
||||
find_package(Git)
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short=7 HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_SHORT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=7
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
endif()
|
||||
else()
|
||||
set(GIT_SHORT_HASH "${OVERRIDE_VERSION}")
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/include/git_version.h.in
|
||||
${CMAKE_BINARY_DIR}/generated/git_version.h)
|
||||
|
||||
include_directories(${CMAKE_BINARY_DIR}/generated)
|
||||
@@ -39,11 +74,9 @@ add_compile_options($<$<COMPILE_LANGUAGE:CXX>:-fno-strict-aliasing> $<$<COMPILE_
|
||||
|
||||
add_subdirectory(Source/)
|
||||
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
endif()
|
||||
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
|
||||
if (BUILD_TESTING)
|
||||
add_subdirectory(unittests/)
|
||||
|
||||
@@ -156,7 +156,7 @@ def print_man_environment_tail():
|
||||
"APP_CONFIG_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for configuration files",
|
||||
"By default FEX will look in ${XDG_CONFIG_HOME, $HOME/.config}/fex-emu/",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEX: Relative to the FEX binary",
|
||||
@@ -168,7 +168,7 @@ def print_man_environment_tail():
|
||||
"APP_CONFIG",
|
||||
[
|
||||
"Allows the user to override where FEX looks for only the application config file",
|
||||
"By default FEX will look in ${XDG_CONFIG_HOME, $HOME/.config}/fex-emu/Config.json",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/Config.json",
|
||||
"This will override this file location",
|
||||
"One must be careful with this option as it will override any applications that load with execve as well"
|
||||
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
|
||||
@@ -182,7 +182,7 @@ def print_man_environment_tail():
|
||||
"APP_DATA_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for data files",
|
||||
"By default FEX will look in {$XDG_DATA_HOME, $HOME/.local/share}/fex-emu/",
|
||||
"By default FEX will look in {$HOME, $XDG_DATA_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
"This is the folder where FEX stores generated files like IR cache"
|
||||
],
|
||||
@@ -200,15 +200,6 @@ def print_man_environment_tail():
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"APP_CACHE_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX stores and loads cache files",
|
||||
"By default FEX will look in ${XDG_CACHE_HOME, $HOME/.cache}/fex-emu/",
|
||||
"This will override the full path, trailing forward-slash is expected to exist",
|
||||
],
|
||||
"''", True)
|
||||
|
||||
def print_man_header():
|
||||
header ='''.Dd {0}
|
||||
.Dt FEX
|
||||
@@ -234,7 +225,7 @@ FEX is very much work in progress, so expect things to change.
|
||||
def print_man_tail():
|
||||
tail ='''.Sh FILES
|
||||
.Bl -tag -width "$prefix/share/fex-emu/GuestThunks" -compact
|
||||
.It Pa $XDG_CONFIG_DIR/fex-emu
|
||||
.It Pa $XDG_HOME_DIR/.fex-emu
|
||||
Default FEX user configuration directory
|
||||
.It Pa $prefix/share/fex-emu/AppConfig
|
||||
System level application configuration files
|
||||
@@ -407,32 +398,6 @@ def print_parse_enum_options(options):
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_affects_codegen_options(options, unnamed_options):
|
||||
output_argloader.write("#ifdef CONFIG_AFFECTSCODEGEN\n")
|
||||
output_argloader.write("#undef CONFIG_AFFECTSCODEGEN\n")
|
||||
|
||||
TotalConfigOptions = 0
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
TotalConfigOptions += 1
|
||||
for op_group, group_vals in unnamed_options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
TotalConfigOptions += 1
|
||||
|
||||
output_argloader.write("constexpr static std::array<bool, {}> Config_AffectsCodeGen = {{{{\n".format(TotalConfigOptions))
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
assert "AffectsCodeGen" in op_vals, "All config options must be marked if they affect codegen."
|
||||
output_argloader.write("\t{}, // {}\n".format(op_vals["AffectsCodeGen"], op_key))
|
||||
|
||||
for op_group, group_vals in unnamed_options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
assert "AffectsCodeGen" in op_vals, "All config options must be marked if they affect codegen."
|
||||
output_argloader.write("\t{}, // {}\n".format(op_vals["AffectsCodeGen"], op_key))
|
||||
output_argloader.write("}};\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
if (len(sys.argv) < 5):
|
||||
sys.exit()
|
||||
|
||||
@@ -477,6 +442,4 @@ print_parse_jsonloader_options(options);
|
||||
# Generate enum variable options
|
||||
print_parse_enum_options(options);
|
||||
|
||||
print_affects_codegen_options(options, unnamed_options);
|
||||
|
||||
output_argloader.close()
|
||||
@@ -58,10 +58,10 @@ class OpDefinition:
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
TiedSource: int
|
||||
Inline: list[str]
|
||||
Arguments: list[OpArgument]
|
||||
EmitValidation: list[str]
|
||||
Desc: list[str]
|
||||
Inline: list
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
|
||||
def __init__(self):
|
||||
self.Name = None
|
||||
@@ -92,14 +92,19 @@ class OpDefinition:
|
||||
attrs = vars(self)
|
||||
print(", ".join("%s: %s" % item for item in attrs.items()))
|
||||
|
||||
IRTypesToCXX: dict[str, IRType] = {}
|
||||
CXXTypeToIR: dict[str, IRType] = {}
|
||||
IROps: list[OpDefinition] = []
|
||||
IRTypesToCXX = {}
|
||||
CXXTypeToIR = {}
|
||||
IROps = []
|
||||
|
||||
IROpNameSet: set[str] = set()
|
||||
IROpNameMap = {}
|
||||
|
||||
def is_ssa_type(op_type: str):
|
||||
return op_type in {"SSA", "GPR", "GPRPair", "FPR"}
|
||||
def is_ssa_type(type):
|
||||
if (type == "SSA" or
|
||||
type == "GPR" or
|
||||
type == "GPRPair" or
|
||||
type == "FPR"):
|
||||
return True
|
||||
return False
|
||||
|
||||
def parse_irtypes(irtypes):
|
||||
for op_key, op_val in irtypes.items():
|
||||
@@ -214,8 +219,11 @@ def parse_ops(ops):
|
||||
OpArg.DefaultInitializer = DefaultInit[1][:-1]
|
||||
|
||||
# If SSA type then we can generate validation for this op
|
||||
if OpArg.IsSSA and OpArg.Type in {"GPR", "GPRPair", "FPR"}:
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == RegClass::Invalid || WalkFindRegClass({ArgName}) == RegClass::{OpArg.Type}")
|
||||
if (OpArg.IsSSA and
|
||||
(OpArg.Type == "GPR" or
|
||||
OpArg.Type == "GPRPair" or
|
||||
OpArg.Type == "FPR")):
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
|
||||
|
||||
OpArg.Name = ArgName
|
||||
OpArg.NameWithPrefix = NameWithPrefix
|
||||
@@ -251,10 +259,6 @@ def parse_ops(ops):
|
||||
|
||||
if "Desc" in op_val:
|
||||
OpDef.Desc = op_val["Desc"]
|
||||
if not isinstance(OpDef.Desc, list):
|
||||
ExitError(f"Desc field for op {OpDef.Name} must be an array of strings")
|
||||
if not all(isinstance(item, str) for item in OpDef.Desc):
|
||||
ExitError(f"Desc field for op {OpDef.Name} must only contain strings")
|
||||
|
||||
if "DynamicDispatch" in op_val:
|
||||
OpDef.DynamicDispatch = bool(op_val["DynamicDispatch"])
|
||||
@@ -292,28 +296,21 @@ def parse_ops(ops):
|
||||
#OpDef.print()
|
||||
|
||||
# Error on duplicate op
|
||||
if OpDef.Name in IROpNameSet:
|
||||
if OpDef.Name in IROpNameMap:
|
||||
ExitError("Duplicate Op defined! {}".format(OpDef.Name))
|
||||
|
||||
IROps.append(OpDef)
|
||||
IROpNameSet.add(OpDef.Name)
|
||||
IROpNameMap[OpDef.Name] = 1
|
||||
|
||||
# Print out enum values
|
||||
def print_enums(enums):
|
||||
def print_enums():
|
||||
output_file.write("#ifdef IROP_ENUM\n")
|
||||
output_file.write("enum IROps : uint16_t {\n")
|
||||
|
||||
for op in IROps:
|
||||
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
|
||||
output_file.write("};\n")
|
||||
|
||||
for name, members in enums.items():
|
||||
output_file.write(f"enum {name} {{\n")
|
||||
for member in members:
|
||||
if member:
|
||||
output_file.write(f"\t{member}\n")
|
||||
else:
|
||||
output_file.write("\n")
|
||||
output_file.write("};\n\n")
|
||||
output_file.write("};\n")
|
||||
|
||||
output_file.write("#undef IROP_ENUM\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -411,7 +408,7 @@ def print_ir_sizes():
|
||||
[[nodiscard, gnu::const]] std::string_view const& GetName(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetRAArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] FEXCore::IR::RegClass GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool HasSideEffects(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool ImplicitFlagClobber(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool GetHasDest(IROps Op);
|
||||
@@ -425,29 +422,30 @@ def print_ir_sizes():
|
||||
def print_ir_reg_classes():
|
||||
output_file.write("#ifdef IROP_REG_CLASSES_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<FEXCore::IR::RegClass, IROps::OP_LAST + 1> IRRegClasses = {\n")
|
||||
output_file.write("constexpr std::array<FEXCore::IR::RegisterClassType, IROps::OP_LAST + 1> IRRegClasses = {\n")
|
||||
for op in IROps:
|
||||
if op.Name == "Last":
|
||||
output_file.write("\tRegClass::Invalid,\n")
|
||||
output_file.write("\tFEXCore::IR::InvalidClass,\n")
|
||||
else:
|
||||
if op.HasDest and op.DestType is None:
|
||||
Class = "Invalid"
|
||||
if op.HasDest and op.DestType == None:
|
||||
ExitError("IR op {} has destination with no destination class".format(op.Name))
|
||||
|
||||
if op.HasDest and op.DestType == "SSA": # Special case SSA type
|
||||
output_file.write("\tRegClass::Complex,\n")
|
||||
output_file.write("\tFEXCore::IR::ComplexClass,\n")
|
||||
elif op.HasDest:
|
||||
output_file.write("\tRegClass::{},\n".format(op.DestType))
|
||||
output_file.write("\tFEXCore::IR::{}Class,\n".format(op.DestType))
|
||||
else:
|
||||
# No destination so it has an invalid destination class
|
||||
output_file.write("\tRegClass::Invalid, // No destination\n")
|
||||
output_file.write("\tFEXCore::IR::InvalidClass, // No destination\n")
|
||||
|
||||
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("// Make sure our array maps directly to the IROps enum\n")
|
||||
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == RegClass::Invalid);\n\n")
|
||||
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == FEXCore::IR::InvalidClass);\n\n")
|
||||
|
||||
output_file.write("FEXCore::IR::RegClass GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
|
||||
output_file.write("FEXCore::IR::RegisterClassType GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
|
||||
|
||||
output_file.write("#undef IROP_REG_CLASSES_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -570,7 +568,9 @@ def print_ir_arg_printer():
|
||||
|
||||
SSAArgNum = 0
|
||||
FirstArg = True
|
||||
for arg in op.Arguments:
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
|
||||
# No point printing temporaries that we can't recover
|
||||
if arg.Temporary:
|
||||
continue
|
||||
@@ -607,100 +607,100 @@ def print_validation(op):
|
||||
def print_ir_allocator_helpers():
|
||||
output_file.write("#ifdef IROP_ALLOCATE_HELPERS\n")
|
||||
|
||||
output_file.write("\ttemplate <class T>\n"
|
||||
"\tstruct Wrapper final {\n"
|
||||
"\t\tT *first;\n"
|
||||
"\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n"
|
||||
"\n"
|
||||
"\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n"
|
||||
"\t\toperator OrderedNode *() { return Node; }\n"
|
||||
"\t\toperator const OrderedNode *() const { return Node; }\n"
|
||||
"\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n"
|
||||
"\t};\n")
|
||||
output_file.write("\ttemplate <class T>\n")
|
||||
output_file.write("\tstruct Wrapper final {\n")
|
||||
output_file.write("\t\tT *first;\n")
|
||||
output_file.write("\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n")
|
||||
output_file.write("\n")
|
||||
output_file.write("\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n")
|
||||
output_file.write("\t\toperator OrderedNode *() { return Node; }\n")
|
||||
output_file.write("\t\toperator const OrderedNode *() const { return Node; }\n")
|
||||
output_file.write("\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n")
|
||||
output_file.write("\t};\n")
|
||||
|
||||
output_file.write("\ttemplate <class T>\n"
|
||||
"\tusing IRPair = Wrapper<T>;\n\n")
|
||||
output_file.write("\ttemplate <class T>\n")
|
||||
output_file.write("\tusing IRPair = Wrapper<T>;\n\n")
|
||||
|
||||
output_file.write("\tIRPair<IROp_Header> AllocateRawOp(size_t HeaderSize) {\n"
|
||||
"\t\tauto Op = reinterpret_cast<IROp_Header*>(DualListData.DataAllocate(HeaderSize));\n"
|
||||
"\t\tmemset(Op, 0, HeaderSize);\n"
|
||||
"\t\tOp->Op = IROps::OP_DUMMY;\n"
|
||||
"\t\treturn IRPair<IROp_Header>{Op, CreateNode(Op)};\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\tIRPair<IROp_Header> AllocateRawOp(size_t HeaderSize) {\n")
|
||||
output_file.write("\t\tauto Op = reinterpret_cast<IROp_Header*>(DualListData.DataAllocate(HeaderSize));\n")
|
||||
output_file.write("\t\tmemset(Op, 0, HeaderSize);\n")
|
||||
output_file.write("\t\tOp->Op = IROps::OP_DUMMY;\n")
|
||||
output_file.write("\t\treturn IRPair<IROp_Header>{Op, CreateNode(Op)};\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\ttemplate<class T, IROps T2>\n"
|
||||
"\tT *AllocateOrphanOp() {\n"
|
||||
"\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n"
|
||||
"\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n"
|
||||
"\t\tmemset(Op, 0, Size);\n"
|
||||
"\t\tOp->Header.Op = T2;\n"
|
||||
"\t\treturn Op;\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\ttemplate<class T, IROps T2>\n")
|
||||
output_file.write("\tT *AllocateOrphanOp() {\n")
|
||||
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
|
||||
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
|
||||
output_file.write("\t\tmemset(Op, 0, Size);\n")
|
||||
output_file.write("\t\tOp->Header.Op = T2;\n")
|
||||
output_file.write("\t\treturn Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\ttemplate<class T, IROps T2>\n"
|
||||
"\tIRPair<T> AllocateOp() {\n"
|
||||
"\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n"
|
||||
"\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n"
|
||||
"\t\tmemset(Op, 0, Size);\n"
|
||||
"\t\tOp->Header.Op = T2;\n"
|
||||
"\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\ttemplate<class T, IROps T2>\n")
|
||||
output_file.write("\tIRPair<T> AllocateOp() {\n")
|
||||
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
|
||||
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
|
||||
output_file.write("\t\tmemset(Op, 0, Size);\n")
|
||||
output_file.write("\t\tOp->Header.Op = T2;\n")
|
||||
output_file.write("\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n"
|
||||
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
|
||||
"\t\treturn HeaderOp->Size;\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->Size;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n"
|
||||
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
|
||||
"\t\treturn HeaderOp->ElementSize;\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->ElementSize;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n"
|
||||
"\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n"
|
||||
"\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n")
|
||||
output_file.write("\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n"
|
||||
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
|
||||
"\t\treturn GetHasDest(HeaderOp->Op);\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn GetHasDest(HeaderOp->Op);\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tIROps GetOpType(const OrderedNode *Op) const {\n"
|
||||
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
|
||||
"\t\treturn HeaderOp->Op;\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\tIROps GetOpType(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n"
|
||||
"\t\treturn GetRegClass(GetOpType(Op));\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\tFEXCore::IR::RegisterClassType GetOpRegClass(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tstd::string_view const& GetOpName(const OrderedNode *Op) const {\n"
|
||||
"\t\treturn IR::GetName(GetOpType(Op));\n"
|
||||
"\t}\n\n")
|
||||
output_file.write("\tstd::string_view const& GetOpName(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\treturn IR::GetName(GetOpType(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
# Generate helpers with operands
|
||||
for op in IROps:
|
||||
if op.Name != "Last":
|
||||
output_file.write("\t///\n".join(["\t/// {}\n" .format(comment) for comment in op.Desc]))
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
# Output SSA args first
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
elif arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("OrderedNodeWrapper {}".format(arg.Name))
|
||||
else:
|
||||
# User defined op that is stored
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
|
||||
if arg.DefaultInitializer:
|
||||
if arg.DefaultInitializer != None:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
@@ -756,22 +756,22 @@ def print_ir_allocator_helpers():
|
||||
|
||||
# Now do the OrderedNode * version if necessary
|
||||
if op.SSAArgNum:
|
||||
output_file.write("\t///\n".join(["\t/// {}\n" .format(comment) for comment in op.Desc]))
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
elif arg.IsSSA:
|
||||
output_file.write("OrderedNode *{}".format(arg.Name))
|
||||
else:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
|
||||
if arg.DefaultInitializer:
|
||||
if arg.DefaultInitializer != None:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
@@ -812,15 +812,16 @@ def print_ir_allocator_helpers():
|
||||
print_validation(op)
|
||||
|
||||
output_file.write(f"\t\treturn _{op.Name}(")
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
output_file.write(arg.Name)
|
||||
if arg.IsSSA:
|
||||
output_file.write("->Wrapped(ListDataBegin)")
|
||||
if not LastArg:
|
||||
output_file.write(", ")
|
||||
output_file.write(");\n")
|
||||
output_file.write("\t}\n\n")
|
||||
output_file.write(");\n");
|
||||
output_file.write("\t}\n\n");
|
||||
|
||||
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
@@ -851,8 +852,8 @@ def print_ir_dispatcher_dispatch():
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
|
||||
if len(sys.argv) < 4:
|
||||
ExitError("Insufficient parameters passed to script")
|
||||
if (len(sys.argv) < 4):
|
||||
ExitError()
|
||||
|
||||
output_filename = sys.argv[2]
|
||||
output_dispatcher_filename = sys.argv[3]
|
||||
@@ -864,7 +865,6 @@ json_file.close()
|
||||
json_object = json.loads(json_text)
|
||||
json_object = {k.upper(): v for k, v in json_object.items()}
|
||||
|
||||
enums = json_object["ENUMS"]
|
||||
ops = json_object["OPS"]
|
||||
irtypes = json_object["IRTYPES"]
|
||||
defines = json_object["DEFINES"]
|
||||
@@ -874,7 +874,7 @@ parse_ops(ops)
|
||||
|
||||
output_file = open(output_filename, "w")
|
||||
|
||||
print_enums(enums)
|
||||
print_enums()
|
||||
print_ir_structs(defines)
|
||||
print_ir_sizes()
|
||||
print_ir_reg_classes()
|
||||
|
||||
@@ -1,31 +1,29 @@
|
||||
set(MAN_DIR share/man CACHE PATH "MAN_DIR")
|
||||
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set(FEXCORE_BASE_SRCS
|
||||
set (FEXCORE_BASE_SRCS
|
||||
Interface/Config/Config.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/FileLoading.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
Utils/SpinWaitLock.cpp
|
||||
Utils/WildcardMatcher.cpp)
|
||||
)
|
||||
|
||||
if (NOT MINGW)
|
||||
if (NOT MINGW_BUILD)
|
||||
list(APPEND FEXCORE_BASE_SRCS
|
||||
Utils/Allocator/64BitAllocator.cpp)
|
||||
endif()
|
||||
|
||||
set(SRCS
|
||||
set (SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/DiskCache.cpp
|
||||
Interface/Core/CodeCache.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/Addressing.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/SharedCodeBufferManager.cpp
|
||||
Interface/Core/OpcodeDispatcher/AVX_128.cpp
|
||||
Interface/Core/OpcodeDispatcher/Crypto.cpp
|
||||
Interface/Core/OpcodeDispatcher/Flags.cpp
|
||||
@@ -33,6 +31,7 @@ set(SRCS
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
@@ -70,10 +69,10 @@ set(SRCS
|
||||
Utils/LongJump.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/WorkQueueThread.cpp
|
||||
Utils/Profiler.cpp)
|
||||
Utils/Profiler.cpp
|
||||
)
|
||||
|
||||
if (ARCHITECTURE_arm64)
|
||||
if (_M_ARM_64)
|
||||
list(APPEND SRCS Utils/ArchHelpers/Arm64.cpp)
|
||||
else()
|
||||
list(APPEND SRCS Utils/ArchHelpers/Arm64_stubs.cpp)
|
||||
@@ -86,54 +85,42 @@ endif()
|
||||
|
||||
set(DEFINES -DJIT_ARM64)
|
||||
|
||||
if (ARCHITECTURE_x86_64)
|
||||
list(APPEND DEFINES -DARCHITECTURE_x86_64=1)
|
||||
if (_M_X86_64)
|
||||
list(APPEND DEFINES -D_M_X86_64=1)
|
||||
endif()
|
||||
|
||||
if (ARCHITECTURE_arm64)
|
||||
list(APPEND DEFINES -DARCHITECTURE_arm64=1)
|
||||
if (_M_ARM_64)
|
||||
list(APPEND DEFINES -D_M_ARM_64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_VIXL_DISASSEMBLER)
|
||||
list(APPEND DEFINES -DVIXL_DISASSEMBLER=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ZYDIS)
|
||||
list(APPEND DEFINES -DZYDIS_DISASSEMBLER=1)
|
||||
endif()
|
||||
|
||||
if (ARCHITECTURE_arm64 AND HAS_CLANG_PRESERVE_ALL)
|
||||
if (_M_ARM_64 AND HAS_CLANG_PRESERVE_ALL)
|
||||
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=__attribute__((preserve_all));-DFEXCORE_HAS_PRESERVE_ALL_ATTR=1")
|
||||
else()
|
||||
list(APPEND DEFINES "-DFEXCORE_PRESERVE_ALL_ATTR=;-DFEXCORE_HAS_PRESERVE_ALL_ATTR=0")
|
||||
endif()
|
||||
|
||||
set(LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter cephes_128bit)
|
||||
set (LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter cephes_128bit)
|
||||
|
||||
if (ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
list(APPEND LIBS vixl::vixl)
|
||||
list (APPEND LIBS vixl)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ZYDIS)
|
||||
list(APPEND LIBS Zydis::Zydis)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW)
|
||||
list(APPEND LIBS dl)
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
else()
|
||||
list(APPEND LIBS synchronization)
|
||||
if (ARCHITECTURE_arm64ec)
|
||||
list(APPEND LIBS mincore)
|
||||
list (APPEND LIBS synchronization)
|
||||
if (_M_ARM_64EC)
|
||||
list (APPEND LIBS mincore)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
# GCC requires libatomic to use 128-bit atomics
|
||||
list(APPEND LIBS atomic)
|
||||
endif()
|
||||
|
||||
# Generate config
|
||||
configure_file(${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
${CMAKE_BINARY_DIR}/generated/Config/Config.json)
|
||||
|
||||
# Generate IR include file
|
||||
@@ -148,10 +135,11 @@ add_custom_command(
|
||||
OUTPUT "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
|
||||
"${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}")
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
|
||||
)
|
||||
|
||||
set_source_files_properties(${OUTPUT_NAME} PROPERTIES GENERATED TRUE)
|
||||
set_source_files_properties(${OUTPUT_NAME} PROPERTIES
|
||||
GENERATED TRUE)
|
||||
|
||||
# Generate IR documentation
|
||||
set(OUTPUT_IR_DOC "${CMAKE_BINARY_DIR}/IR.md")
|
||||
@@ -160,10 +148,11 @@ add_custom_command(
|
||||
OUTPUT "${OUTPUT_IR_DOC}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
|
||||
"${INPUT_NAME}" "${OUTPUT_IR_DOC}")
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py" "${INPUT_NAME}" "${OUTPUT_IR_DOC}"
|
||||
)
|
||||
|
||||
set_source_files_properties(${OUTPUT_IR_NAME} PROPERTIES GENERATED TRUE)
|
||||
set_source_files_properties(${OUTPUT_IR_NAME} PROPERTIES
|
||||
GENERATED TRUE)
|
||||
|
||||
# Create the target
|
||||
add_custom_target(IR_INC
|
||||
@@ -187,12 +176,14 @@ add_custom_command(
|
||||
DEPENDS "${INPUT_CONFIG_NAME}"
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py" "${INPUT_CONFIG_NAME}" "${OUTPUT_CONFIG_NAME}" "${OUTPUT_MAN_NAME}"
|
||||
"${OUTPUT_CONFIG_OPTION_NAME}")
|
||||
"${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
)
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_MAN_NAME_COMPRESS}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME}"
|
||||
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}")
|
||||
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}"
|
||||
)
|
||||
|
||||
set_source_files_properties(${OUTPUT_CONFIG_NAME} PROPERTIES
|
||||
GENERATED TRUE)
|
||||
@@ -211,10 +202,8 @@ add_custom_target(CONFIG_INC
|
||||
DEPENDS "${OUTPUT_MAN_NAME}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
|
||||
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
|
||||
endif()
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
|
||||
|
||||
# Add in diagnostic colours if the option is available.
|
||||
# Ninja code generator will kill colours if this isn't here
|
||||
@@ -236,7 +225,8 @@ function(AddDefaultOptionsToTarget Name)
|
||||
target_compile_definitions(${Name} PRIVATE ${DEFINES})
|
||||
add_dependencies(${Name} CONFIG_INC IR_INC)
|
||||
|
||||
target_compile_options(${Name} PRIVATE
|
||||
target_compile_options(${Name}
|
||||
PRIVATE
|
||||
-Wall
|
||||
-Werror=cast-qual
|
||||
-Werror=ignored-qualifiers
|
||||
@@ -244,73 +234,82 @@ function(AddDefaultOptionsToTarget Name)
|
||||
|
||||
-Wno-trigraphs
|
||||
-ffunction-sections
|
||||
-fwrapv)
|
||||
-fwrapv
|
||||
)
|
||||
|
||||
if (GCC_COLOR)
|
||||
target_compile_options(${Name} PRIVATE "-fdiagnostics-color=always")
|
||||
target_compile_options(${Name}
|
||||
PRIVATE
|
||||
"-fdiagnostics-color=always")
|
||||
endif()
|
||||
|
||||
if (CLANG_COLOR)
|
||||
target_compile_options(${Name} PRIVATE "-fcolor-diagnostics")
|
||||
target_compile_options(${Name}
|
||||
PRIVATE
|
||||
"-fcolor-diagnostics")
|
||||
endif()
|
||||
|
||||
LinkerGC(${Name})
|
||||
target_link_libraries(${Name} PUBLIC unordered_dense::unordered_dense)
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
target_link_options(${Name}
|
||||
PRIVATE
|
||||
"LINKER:--gc-sections"
|
||||
"LINKER:--strip-all"
|
||||
"LINKER:--as-needed"
|
||||
)
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
# Build FEXCore_Base static library
|
||||
# Build FEXCore_Config static library
|
||||
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base PUBLIC ${LIBS})
|
||||
target_link_libraries(FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
target_link_libraries(FEXCore_Base PUBLIC TracyClient)
|
||||
target_link_libraries(FEXCore_Base TracyClient)
|
||||
endif()
|
||||
|
||||
function(AddObject Name)
|
||||
add_library(${Name} OBJECT ${SRCS})
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
|
||||
target_link_libraries(${Name} PRIVATE FEXCore_Base)
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
|
||||
# During generation of the import library (dll.a), MinGW needs some extra symbols from libraries
|
||||
# such as fmt, which are propagated by FEXCore_Base. Wonderful.
|
||||
if (MINGW)
|
||||
target_link_libraries(${Name} PRIVATE FEXCore_Base)
|
||||
endif()
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
endfunction()
|
||||
|
||||
AddObject(${PROJECT_NAME}_object)
|
||||
AddObject(${PROJECT_NAME}_object OBJECT)
|
||||
AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
if (NOT MINGW AND NOT BUILD_STEAM_SUPPORT)
|
||||
install(TARGETS ${PROJECT_NAME}_shared LIBRARY
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
COMPONENT Libraries)
|
||||
if (NOT MINGW_BUILD)
|
||||
install(TARGETS ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
COMPONENT Libraries)
|
||||
endif()
|
||||
|
||||
# Meta-library to link jemalloc libraries enabled in the build configuration.
|
||||
# Only needed for targets that run emulation. For others, use JemallocDummy.
|
||||
add_library(JemallocLibs STATIC Utils/AllocatorHooks.cpp)
|
||||
if (ENABLE_FEX_ALLOCATOR)
|
||||
target_compile_definitions(JemallocLibs PRIVATE ENABLE_FEX_ALLOCATOR=1)
|
||||
target_link_libraries(JemallocLibs PUBLIC rpmalloc)
|
||||
target_include_directories(JemallocLibs PRIVATE "${PROJECT_SOURCE_DIR}/include/")
|
||||
if (ENABLE_JEMALLOC)
|
||||
target_compile_definitions(JemallocLibs PRIVATE ENABLE_JEMALLOC=1 JEMALLOC_NO_RENAME=1)
|
||||
target_link_libraries(JemallocLibs PUBLIC FEX_jemalloc)
|
||||
endif()
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
set_source_files_properties(Interface/HLE/Thunks/Thunks.cpp PROPERTIES COMPILE_DEFINITIONS ENABLE_JEMALLOC_GLIBC=1)
|
||||
target_link_libraries(JemallocLibs INTERFACE FEX_jemalloc_glibc)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW)
|
||||
if (NOT MINGW_BUILD)
|
||||
# Dummy project to use for host tools.
|
||||
# This overrides use of jemalloc in FEXCore with the normal glibc allocator.
|
||||
add_library(JemallocDummy STATIC Utils/AllocatorHooks.cpp)
|
||||
@@ -318,4 +317,4 @@ if (NOT MINGW)
|
||||
endif()
|
||||
|
||||
# The shared library should always link enabled jemalloc libraries
|
||||
target_link_libraries(${PROJECT_NAME}_shared PRIVATE JemallocLibs)
|
||||
target_link_libraries(${PROJECT_NAME}_shared JemallocLibs)
|
||||
@@ -18,7 +18,7 @@ struct BitSet final {
|
||||
constexpr static size_t MinimumSize = sizeof(ElementType);
|
||||
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
|
||||
|
||||
ElementType* Memory {};
|
||||
ElementType* Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = ToBytes(Elements);
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
@@ -33,15 +33,14 @@ struct BitSet final {
|
||||
FEXCore::Allocator::free(Memory);
|
||||
Memory = nullptr;
|
||||
}
|
||||
[[nodiscard]]
|
||||
bool Get(T Element) const {
|
||||
bool Get(T Element) {
|
||||
return (Memory[Element / MinimumSizeBits] & (1ULL << (Element % MinimumSizeBits))) != 0;
|
||||
}
|
||||
void Set(T Element) {
|
||||
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void Clear(T Element) {
|
||||
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
|
||||
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void MemClear(size_t Elements) {
|
||||
memset(Memory, 0, ToBytes(Elements));
|
||||
@@ -49,15 +48,13 @@ struct BitSet final {
|
||||
void MemSet(size_t Elements) {
|
||||
memset(Memory, 0xFF, ToBytes(Elements));
|
||||
}
|
||||
[[nodiscard]]
|
||||
static size_t ToBytes(size_t Elements) {
|
||||
return AlignUp(Elements, MinimumSizeBits) / 8;
|
||||
uint32_t ToBytes(size_t Elements) {
|
||||
return AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
}
|
||||
|
||||
// This very explicitly doesn't let you take an address
|
||||
// Is only a getter
|
||||
[[nodiscard]]
|
||||
bool operator[](T Element) const {
|
||||
bool operator[](T Element) {
|
||||
return Get(Element);
|
||||
}
|
||||
};
|
||||
@@ -65,37 +62,35 @@ struct BitSet final {
|
||||
template<typename T>
|
||||
struct BitSetView final {
|
||||
using ElementType = T;
|
||||
constexpr static size_t MinimumSize = BitSet<T>::MinimumSize;
|
||||
constexpr static size_t MinimumSizeBits = BitSet<T>::MinimumSizeBits;
|
||||
constexpr static size_t MinimumSize = sizeof(ElementType);
|
||||
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
|
||||
|
||||
ElementType* Memory {};
|
||||
ElementType* Memory;
|
||||
|
||||
void GetView(BitSet<T>& Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool Get(T Element) const {
|
||||
bool Get(T Element) {
|
||||
return (Memory[Element / MinimumSizeBits] & (1ULL << (Element % MinimumSizeBits))) != 0;
|
||||
}
|
||||
void Set(T Element) {
|
||||
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void Clear(T Element) {
|
||||
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
|
||||
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void MemClear(size_t Elements) {
|
||||
memset(Memory, 0, BitSet<T>::ToBytes(Elements));
|
||||
memset(Memory, 0, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
}
|
||||
void MemSet(size_t Elements) {
|
||||
memset(Memory, 0xFF, BitSet<T>::ToBytes(Elements));
|
||||
memset(Memory, 0xFF, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
}
|
||||
|
||||
// This very explicitly doesn't let you take an address
|
||||
// Is only a getter
|
||||
[[nodiscard]]
|
||||
bool operator[](T Element) const {
|
||||
bool operator[](T Element) {
|
||||
return Get(Element);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
@@ -13,7 +12,7 @@ namespace FEXCore {
|
||||
// Buffered JIT symbol tracking.
|
||||
struct JITSymbolBuffer {
|
||||
// Maximum buffer size to ensure we are a page in size.
|
||||
constexpr static size_t BUFFER_SIZE = FEXCore::Utils::FEX_PAGE_SIZE - (8 * 2);
|
||||
constexpr static size_t BUFFER_SIZE = 4096 - (8 * 2);
|
||||
// Maximum distance until the end of the buffer to do a write.
|
||||
constexpr static size_t NEEDS_WRITE_DISTANCE = BUFFER_SIZE - 64;
|
||||
// Maximum time threshhold to wait before a buffer write occurs.
|
||||
@@ -28,7 +27,7 @@ struct JITSymbolBuffer {
|
||||
size_t Offset {};
|
||||
char Buffer[BUFFER_SIZE] {};
|
||||
};
|
||||
static_assert(sizeof(JITSymbolBuffer) == FEXCore::Utils::FEX_PAGE_SIZE, "Ensure this is one page in size");
|
||||
static_assert(sizeof(JITSymbolBuffer) == 4096, "Ensure this is one page in size");
|
||||
|
||||
class JITSymbols final {
|
||||
public:
|
||||
|
||||
@@ -4,9 +4,9 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/BitUtils.h>
|
||||
#include "cephes_128bit.h"
|
||||
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
@@ -19,7 +19,7 @@ extern "C" {
|
||||
}
|
||||
|
||||
struct FEX_PACKED X80SoftFloat {
|
||||
#ifdef ARCHITECTURE_x86_64
|
||||
#ifdef _M_X86_64
|
||||
// Define this to push some operations to x87
|
||||
// Only useful to see if precision loss is killing something
|
||||
// #define DEBUG_X86_FLOAT
|
||||
@@ -30,33 +30,29 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#define BIGFLOAT float128_t
|
||||
#define BIGFLOATSIZE 16
|
||||
#endif
|
||||
#elif defined(ARCHITECTURE_arm64)
|
||||
#elif defined(_M_ARM_64)
|
||||
#define BIGFLOAT float128_t
|
||||
#define BIGFLOATSIZE 16
|
||||
#else
|
||||
#error No 128bit float for this target!
|
||||
#endif
|
||||
|
||||
uint64_t Significand;
|
||||
union {
|
||||
uint16_t Raw;
|
||||
struct {
|
||||
uint16_t Exponent : 15;
|
||||
uint16_t Sign : 1;
|
||||
};
|
||||
} Top;
|
||||
uint64_t Significand : 64;
|
||||
uint16_t Exponent : 15;
|
||||
uint16_t Sign : 1;
|
||||
|
||||
X80SoftFloat() {
|
||||
memset(this, 0, sizeof(*this));
|
||||
}
|
||||
X80SoftFloat(uint16_t _Sign, uint16_t _Exponent, uint64_t _Significand)
|
||||
: Significand {_Significand}
|
||||
, Top {.Raw = static_cast<uint16_t>((_Exponent & 0x7FFF) | (_Sign << 15))} {}
|
||||
, Exponent {_Exponent}
|
||||
, Sign {_Sign} {}
|
||||
|
||||
fextl::string str() const {
|
||||
fextl::ostringstream string;
|
||||
string << std::hex << Top.Sign;
|
||||
string << "_" << Top.Exponent;
|
||||
string << std::hex << Sign;
|
||||
string << "_" << Exponent;
|
||||
string << "_" << (Significand >> 63);
|
||||
string << "_" << (Significand & ((1ULL << 63) - 1));
|
||||
return string.str();
|
||||
@@ -167,18 +163,18 @@ struct FEX_PACKED X80SoftFloat {
|
||||
X80SoftFloat result = 0;
|
||||
if (HandleInfinityOp(state, lhs, result)) {
|
||||
return result;
|
||||
} else if (lhs.Top.Exponent == 0x7FFF && (lhs.Significand & 0x7FFFFFFFFFFFFFFFULL)) { // NaN
|
||||
} else if (lhs.Exponent == 0x7FFF && (lhs.Significand & 0x7FFFFFFFFFFFFFFFULL)) { // NaN
|
||||
// propagate NaN
|
||||
state->exceptionFlags |= softfloat_flag_invalid;
|
||||
return lhs;
|
||||
}
|
||||
|
||||
// Check for zero divisor - fprem(x, 0) is invalid operation
|
||||
if (rhs.Top.Exponent == 0 && rhs.Significand == 0) {
|
||||
if (rhs.Exponent == 0 && rhs.Significand == 0) {
|
||||
state->exceptionFlags |= softfloat_flag_invalid;
|
||||
// Return QNaN
|
||||
result.Top.Sign = 0;
|
||||
result.Top.Exponent = 0x7FFF;
|
||||
result.Sign = 0;
|
||||
result.Exponent = 0x7FFF;
|
||||
result.Significand = 0xC000000000000000ULL;
|
||||
return result;
|
||||
}
|
||||
@@ -257,16 +253,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
return Result;
|
||||
#else
|
||||
// Zero is a special case, the significand for +/- 0 is +/- zero.
|
||||
if (lhs.Top.Exponent == 0x0 && lhs.Significand == 0x0) {
|
||||
return lhs;
|
||||
}
|
||||
// Inf/NaN pass through unchanged in the significand slot.
|
||||
if (lhs.Top.Exponent == 0x7FFF) {
|
||||
if (lhs.Exponent == 0x0 && lhs.Significand == 0x0) {
|
||||
return lhs;
|
||||
}
|
||||
X80SoftFloat Tmp = lhs;
|
||||
Tmp.Top.Exponent = 0x3FFF;
|
||||
Tmp.Top.Sign = lhs.Top.Sign;
|
||||
Tmp.Exponent = 0x3FFF;
|
||||
Tmp.Sign = lhs.Sign;
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
@@ -288,20 +280,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
return Result;
|
||||
#else
|
||||
// Zero is a special case, the exponent is always -inf
|
||||
if (lhs.Top.Exponent == 0x0 && lhs.Significand == 0x0) {
|
||||
if (lhs.Exponent == 0x0 && lhs.Significand == 0x0) {
|
||||
X80SoftFloat Result(1, 0x7FFFUL, 0x8000'0000'0000'0000UL);
|
||||
return Result;
|
||||
}
|
||||
// +/-Inf returns +Inf in the exponent slot; NaN propagates.
|
||||
if (lhs.Top.Exponent == 0x7FFF) {
|
||||
if ((lhs.Significand & 0x7FFFFFFFFFFFFFFFULL) == 0) {
|
||||
X80SoftFloat Result(0, 0x7FFFUL, 0x8000'0000'0000'0000UL);
|
||||
return Result;
|
||||
}
|
||||
return lhs;
|
||||
}
|
||||
|
||||
int32_t TrueExp = lhs.Top.Exponent - ExponentBias;
|
||||
int32_t TrueExp = lhs.Exponent - ExponentBias;
|
||||
return i32_to_extF80(TrueExp);
|
||||
#endif
|
||||
}
|
||||
@@ -336,13 +320,6 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#else
|
||||
extFloat80_t Zero {0, 0};
|
||||
if (extF80_eq(state, lhs, Zero)) {
|
||||
// FSCALE(0, +Inf) is 0 * Inf, which is invalid. FSCALE(0, anything
|
||||
// else) is still 0.
|
||||
if (rhs.Top.Exponent == 0x7FFF && rhs.Top.Sign == 0 && (rhs.Significand & 0x7FFFFFFFFFFFFFFFULL) == 0) {
|
||||
state->exceptionFlags |= softfloat_flag_invalid;
|
||||
X80SoftFloat QNaN(0, 0x7FFFUL, 0xC000000000000000ULL);
|
||||
return QNaN;
|
||||
}
|
||||
return lhs;
|
||||
}
|
||||
X80SoftFloat Int = FRNDINT(state, rhs, softfloat_round_minMag);
|
||||
@@ -524,12 +501,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
float ToF32(softfloat_state* state) const {
|
||||
const float32_t Result = extF80_to_f32(state, *this);
|
||||
return std::bit_cast<float>(Result);
|
||||
return FEXCore::BitCast<float>(Result);
|
||||
}
|
||||
|
||||
double ToF64(softfloat_state* state) const {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
return std::bit_cast<double>(Result);
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
}
|
||||
|
||||
FEXCore::VectorRegType ToVector() const {
|
||||
@@ -541,7 +518,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
BIGFLOAT ToFMax(softfloat_state* state) const {
|
||||
#if BIGFLOATSIZE == 16
|
||||
const float128_t Result = extF80_to_f128(state, *this);
|
||||
return std::bit_cast<BIGFLOAT>(Result);
|
||||
return FEXCore::BitCast<BIGFLOAT>(Result);
|
||||
#else
|
||||
BIGFLOAT result {};
|
||||
memcpy(&result, this, sizeof(result));
|
||||
@@ -595,22 +572,23 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
X80SoftFloat(extFloat80_t rhs) {
|
||||
Significand = rhs.signif;
|
||||
Top.Raw = rhs.signExp;
|
||||
Exponent = rhs.signExp & 0x7FFF;
|
||||
Sign = rhs.signExp >> 15;
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, const float rhs) {
|
||||
*this = f32_to_extF80(state, std::bit_cast<float32_t>(rhs));
|
||||
*this = f32_to_extF80(state, FEXCore::BitCast<float32_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, const double rhs) {
|
||||
*this = f64_to_extF80(state, std::bit_cast<float64_t>(rhs));
|
||||
*this = f64_to_extF80(state, FEXCore::BitCast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(state, std::bit_cast<float128_t>(rhs));
|
||||
*this = f128_to_extF80(state, FEXCore::BitCast<float128_t>(rhs));
|
||||
#else
|
||||
*this = std::bit_cast<long double>(rhs);
|
||||
*this = FEXCore::BitCast<long double>(rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -628,7 +606,8 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
void operator=(extFloat80_t rhs) {
|
||||
Significand = rhs.signif;
|
||||
Top.Raw = rhs.signExp;
|
||||
Exponent = rhs.signExp & 0x7FFF;
|
||||
Sign = rhs.signExp >> 15;
|
||||
}
|
||||
|
||||
operator FEXCore::VectorRegType() const {
|
||||
@@ -638,16 +617,16 @@ struct FEX_PACKED X80SoftFloat {
|
||||
operator extFloat80_t() const {
|
||||
extFloat80_t Result {};
|
||||
Result.signif = Significand;
|
||||
Result.signExp = Top.Raw;
|
||||
Result.signExp = Exponent | (Sign << 15);
|
||||
return Result;
|
||||
}
|
||||
|
||||
static bool IsNan(const X80SoftFloat& lhs) {
|
||||
return (lhs.Top.Exponent == 0x7FFF) && (lhs.Significand & IntegerBit) && (lhs.Significand & Bottom62Significand);
|
||||
return (lhs.Exponent == 0x7FFF) && (lhs.Significand & IntegerBit) && (lhs.Significand & Bottom62Significand);
|
||||
}
|
||||
|
||||
static bool SignBit(const X80SoftFloat& lhs) {
|
||||
return lhs.Top.Sign;
|
||||
return lhs.Sign;
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -658,11 +637,11 @@ private:
|
||||
// Helper function to check for infinity and set invalid operation flag.
|
||||
// Returns true if infinity is dealt with, false otherwise.
|
||||
FEXCORE_PRESERVE_ALL_ATTR static bool HandleInfinityOp(softfloat_state* state, const X80SoftFloat& arg, X80SoftFloat& result) {
|
||||
if (arg.Top.Exponent == 0x7FFF && arg.Significand == 0x8000000000000000ULL) {
|
||||
if (arg.Exponent == 0x7FFF && arg.Significand == 0x8000000000000000ULL) {
|
||||
state->exceptionFlags |= softfloat_flag_invalid;
|
||||
// Return QNaN.
|
||||
result.Top.Sign = 0;
|
||||
result.Top.Exponent = 0x7FFF;
|
||||
result.Sign = 0;
|
||||
result.Exponent = 0x7FFF;
|
||||
result.Significand = 0xC000000000000000ULL;
|
||||
return true;
|
||||
}
|
||||
@@ -670,4 +649,9 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
static_assert(sizeof(X80SoftFloat) == 10, "tword must be 10bytes in size");
|
||||
#else
|
||||
// Padding on this extends to 16-bytes rather than 10-bytes on WIN32.
|
||||
static_assert(sizeof(X80SoftFloat) == 16, "tword must be 16bytes in size");
|
||||
#endif
|
||||
@@ -2,24 +2,59 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <concepts>
|
||||
#include <cstdint>
|
||||
#include <string_view>
|
||||
#include <cstdlib>
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::StrConv {
|
||||
template<std::integral T>
|
||||
bool Conv(std::string_view Value, T* Result) {
|
||||
if constexpr (std::is_signed_v<T>) {
|
||||
*Result = static_cast<T>(std::strtoll(Value.data(), nullptr, 0));
|
||||
} else {
|
||||
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
|
||||
}
|
||||
inline bool Conv(std::string_view Value, bool* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename T, typename = std::enable_if_t<std::is_enum_v<T>, T>>
|
||||
bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
|
||||
inline bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int8_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int16_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int32_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int64_t* Result) {
|
||||
*Result = std::strtoll(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
inline bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#ifdef ARCHITECTURE_x86_64
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#include <immintrin.h>
|
||||
#else
|
||||
@@ -13,14 +13,10 @@ struct VectorScalarF64Pair {
|
||||
double val[2];
|
||||
};
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
#ifdef _M_ARM_64
|
||||
// Can't use uint8x16_t directly from arm_neon.h here.
|
||||
// Overrides softfloat-3e's defines which causes problems.
|
||||
#ifdef __clang__
|
||||
using VectorRegType = __attribute__((neon_vector_type(16))) uint8_t;
|
||||
#else
|
||||
using VectorRegType = __attribute__((vector_size(16))) uint8_t;
|
||||
#endif
|
||||
struct VectorRegPairType {
|
||||
VectorRegType val[2];
|
||||
};
|
||||
@@ -29,7 +25,7 @@ static inline VectorRegPairType MakeVectorRegPair(VectorRegType low, VectorRegTy
|
||||
return VectorRegPairType {low, high};
|
||||
}
|
||||
|
||||
#elif defined(ARCHITECTURE_x86_64)
|
||||
#elif defined(_M_X86_64)
|
||||
using VectorRegType = __m128i;
|
||||
using VectorRegPairType = __m256i;
|
||||
|
||||
|
||||
@@ -1,10 +1,9 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/StringConv.h"
|
||||
#include "Utils/Config.h"
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
@@ -31,22 +30,14 @@ class Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Config {
|
||||
namespace detail {
|
||||
namespace DefaultValues {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
|
||||
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
constexpr static std::array<std::string_view, FEXCore::Config::ConfigOption::CONFIG_MAX> option_names = {
|
||||
#define OPT_BASE(type, group, enum, json, default) #json,
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
};
|
||||
} // namespace detail
|
||||
|
||||
std::string_view GetConfigJSONName(FEXCore::Config::ConfigOption option) {
|
||||
return FEXCore::Config::detail::option_names[option];
|
||||
}
|
||||
} // namespace DefaultValues
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR_LOCAL = 0,
|
||||
@@ -56,7 +47,6 @@ enum Paths {
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
PATH_CONFIG_FILE_GLOBAL,
|
||||
PATH_CONFIG_TELEMETRY_FOLDER,
|
||||
PATH_CACHE_DIR,
|
||||
PATH_LAST,
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
@@ -73,10 +63,6 @@ void SetConfigFileLocation(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
void SetCacheDirectory(const std::string_view Path) {
|
||||
Paths[PATH_CACHE_DIR] = Path;
|
||||
}
|
||||
|
||||
const fextl::string& GetTelemetryDirectory() {
|
||||
auto& Path = Paths[PATH_CONFIG_TELEMETRY_FOLDER];
|
||||
if (Path.empty()) {
|
||||
@@ -104,10 +90,6 @@ const fextl::string& GetConfigFileLocation(bool Global) {
|
||||
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
|
||||
}
|
||||
|
||||
const fextl::string& GetCacheDirectory() {
|
||||
return Paths[PATH_CACHE_DIR];
|
||||
}
|
||||
|
||||
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
|
||||
fextl::string ConfigFile = GetConfigDirectory(Global);
|
||||
|
||||
@@ -152,7 +134,7 @@ public:
|
||||
void Load();
|
||||
|
||||
template<typename T>
|
||||
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<StringArrayType, T>)
|
||||
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<DefaultValues::Type::StringArrayType, T>)
|
||||
std::optional<T> GetConv(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
@@ -160,7 +142,7 @@ public:
|
||||
}
|
||||
|
||||
const auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(!std::holds_alternative<StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
LOGMAN_THROW_A_FMT(!std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
if (std::holds_alternative<T>(Value)) [[likely]] {
|
||||
return std::get<T>(Value);
|
||||
@@ -183,7 +165,7 @@ public:
|
||||
|
||||
private:
|
||||
void MergeConfigMap(const LayerOptions& Options);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value);
|
||||
};
|
||||
|
||||
void MetaLayer::Load() {
|
||||
@@ -199,7 +181,7 @@ void MetaLayer::Load() {
|
||||
}
|
||||
|
||||
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value) {
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value) {
|
||||
// Environment variables need a bit of additional work
|
||||
// We want to merge the arrays rather than overwrite entirely
|
||||
auto MetaEnvironment = OptionMap.find(Option);
|
||||
@@ -211,7 +193,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Stri
|
||||
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](const StringArrayType& Value) {
|
||||
const auto AddToMap = [&LookupMap](const DefaultValues::Type::StringArrayType& Value) {
|
||||
for (const auto& EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == fextl::string::npos) {
|
||||
@@ -227,7 +209,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Stri
|
||||
}
|
||||
};
|
||||
|
||||
AddToMap(std::get<StringArrayType>(MetaEnvironment->second));
|
||||
AddToMap(std::get<DefaultValues::Type::StringArrayType>(MetaEnvironment->second));
|
||||
AddToMap(Value);
|
||||
|
||||
// Now with the two layers merged in the map
|
||||
@@ -243,8 +225,8 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto& it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<StringArrayType>(it.second), "Tried to get config of invalid type!");
|
||||
MergeEnvironmentVariables(it.first, std::get<StringArrayType>(it.second));
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(it.second), "Tried to get config of invalid type!");
|
||||
MergeEnvironmentVariables(it.first, std::get<DefaultValues::Type::StringArrayType>(it.second));
|
||||
} else {
|
||||
OptionMap.insert_or_assign(it.first, it.second);
|
||||
}
|
||||
@@ -270,7 +252,7 @@ void Load() {
|
||||
}
|
||||
}
|
||||
|
||||
static fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
|
||||
fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
@@ -325,10 +307,12 @@ constexpr char ContainerManager[] = "/run/host/container-manager";
|
||||
fextl::string FindContainer() {
|
||||
// We only support pressure-vessel at the moment
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::string Manager {};
|
||||
fextl::vector<char> Manager {};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
return FEXCore::StringUtils::Trim(Manager);
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
return ManagerStr;
|
||||
}
|
||||
}
|
||||
return {};
|
||||
@@ -337,10 +321,12 @@ fextl::string FindContainer() {
|
||||
fextl::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::string Manager {};
|
||||
fextl::vector<char> Manager {};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
if (FEXCore::StringUtils::Trim(Manager) == "pressure-vessel") {
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
|
||||
// We are running inside of pressure vessel
|
||||
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
|
||||
return "/run/host/";
|
||||
@@ -437,7 +423,7 @@ bool Exists(ConfigOption Option) {
|
||||
return Meta->OptionExists(Option);
|
||||
}
|
||||
|
||||
std::optional<StringArrayType*> All(ConfigOption Option) {
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
@@ -450,12 +436,6 @@ std::optional<T> GetConv(ConfigOption Option) {
|
||||
return Meta->GetConv<T>(Option);
|
||||
}
|
||||
|
||||
template std::optional<bool> GetConv(ConfigOption Option);
|
||||
template std::optional<uint8_t> GetConv(ConfigOption Option);
|
||||
template std::optional<int32_t> GetConv(ConfigOption Option);
|
||||
template std::optional<uint32_t> GetConv(ConfigOption Option);
|
||||
template std::optional<uint64_t> GetConv(ConfigOption Option);
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
@@ -511,51 +491,13 @@ template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t De
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List) {
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
|
||||
|
||||
#define CONFIG_AFFECTSCODEGEN
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
fextl::string SerializeForCache() {
|
||||
fextl::string Config {};
|
||||
|
||||
auto append_string_triple = [](fextl::string& Config, std::string_view Key, ConfigOption Option, auto Value) {
|
||||
Config.append(Key);
|
||||
Config.append(1, '\0');
|
||||
Config.append(fextl::fmt::format("{}", FEXCore::ToUnderlying(Option)));
|
||||
Config.append(1, '\0');
|
||||
Config.append(fextl::fmt::format("{}", Value));
|
||||
Config.append(1, '\0');
|
||||
};
|
||||
|
||||
const auto SerializeValue = [&Config, append_string_triple]<typename T, ConfigOption Option>(auto ConfigVal, const auto Default) {
|
||||
if (!Config_AffectsCodeGen[FEXCore::ToUnderlying(Option)]) {
|
||||
// Skip everything that the config says doesn't affect codegen.
|
||||
return;
|
||||
}
|
||||
append_string_triple(Config, FEXCore::Config::GetConfigJSONName(Option), Option, ConfigVal());
|
||||
};
|
||||
|
||||
#define OPT_BASE(type, group, enum, json, default) \
|
||||
SerializeValue.template operator()<type, CONFIG_##enum>(FEXCore::Config::Get_##enum(), default);
|
||||
#define OPT_STR(group, enum, json, default) \
|
||||
SerializeValue.template operator()<fextl::string, CONFIG_##enum>(FEXCore::Config::Get_##enum(), default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) // Unsupported.
|
||||
#define OPT_STRENUM(group, enum, json, default) // Unsupported.
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
return Config;
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool CheckConfigMatches(std::string_view Config) {
|
||||
// Serialize current config and just check if it matches.
|
||||
return SerializeForCache() == Config;
|
||||
}
|
||||
|
||||
template void Value<DefaultValues::Type::StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option,
|
||||
DefaultValues::Type::StringArrayType* List);
|
||||
} // namespace FEXCore::Config
|
||||
@@ -4,7 +4,6 @@
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation",
|
||||
"Can cause long JIT compilation times and stutter"
|
||||
@@ -13,40 +12,13 @@
|
||||
"MaxInst": {
|
||||
"Type": "int32",
|
||||
"Default": "5000",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"EnableCodeCachingWIP": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Enable the code caching subsystem"
|
||||
]
|
||||
},
|
||||
"EnableLazyCodeCachingWIP": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enable lazy loading of chunks in code caches"
|
||||
]
|
||||
},
|
||||
"EnableCodeCacheValidation": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enable expensive validation when loading code caches"
|
||||
]
|
||||
},
|
||||
"HostFeatures": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::HostFeatures::OFF",
|
||||
"AffectsCodeGen": "true",
|
||||
"Comment": "Technically affects codegen, but this is serialized elsewhere.",
|
||||
"Enums": {
|
||||
"ENABLESVE": "enablesve",
|
||||
"DISABLESVE": "disablesve",
|
||||
@@ -87,11 +59,7 @@
|
||||
"ENABLEWFXT": "enablewfxt",
|
||||
"DISABLEWFXT": "disablewfxt",
|
||||
"ENABLE3DNOW": "enable3dnow",
|
||||
"DISABLE3DNOW": "disable3dnow",
|
||||
"ENABLESSE4A": "enablesse4a",
|
||||
"DISABLESSE4A": "disablesse4a",
|
||||
"ENABLEMOPS": "enablemops",
|
||||
"DISABLEMOPS": "disablemops"
|
||||
"DISABLE3DNOW": "disable3dnow"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -114,113 +82,35 @@
|
||||
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
|
||||
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it",
|
||||
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it"
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Scales the cycle counter on systems that have low frequencies."
|
||||
]
|
||||
},
|
||||
"HideHybrid": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Hides hybrid CPU core arrangement."
|
||||
]
|
||||
},
|
||||
"CPUFeatureRegisters": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Comment": "Technically affects codegen, but this is serialized in to HostFeatures.",
|
||||
"Desc": [
|
||||
"Allows overriding cpu feature flags for manual testing"
|
||||
]
|
||||
},
|
||||
"DiskCache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables disk caching for code blocks"
|
||||
]
|
||||
},
|
||||
"DiskCacheFileMapping": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Maps cache files for faster reading"
|
||||
]
|
||||
},
|
||||
"DiskCacheValidation": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Debug mode that does nothing but validate code hits"
|
||||
]
|
||||
},
|
||||
"DiskCacheRelocationFilter": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Don't cache blocks with relocations pointing outside of any known region"
|
||||
]
|
||||
},
|
||||
"DiskCacheAnonCaching": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Attempt to cache anonymous code"
|
||||
]
|
||||
},
|
||||
"DiskCachePath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Optional base directory override for disk cache"
|
||||
]
|
||||
},
|
||||
"DiskCacheRODBNames": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Optional list of extra read-only disk cache DBs to consider"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
"RootFS": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Which Root filesystem prefix to use",
|
||||
"This can be a filesystem path",
|
||||
"\teg: ~/RootFS/Debian_x86_64",
|
||||
"Or this can be a name of a rootfs",
|
||||
"If the named rootfs exists in the FEX data folder then it will use that one",
|
||||
"\teg: $XDG_DATA_HOME/fex-emu/RootFS/<RootFS name>/",
|
||||
"If XDG_DATA_HOME is unset, ~/.local/share will be used in its place.",
|
||||
"\teg: $HOME/.local/share/fex-emu/RootFS/<RootFS name>/"
|
||||
"\teg: $HOME/.fex-emu/RootFS/<RootFS name>/",
|
||||
"Or if you have XDG_DATA_HOME the config will search in that directory",
|
||||
"\teg: $XDG_DATA_HOME/.fex-emu/RootFS/<RootFS name>/"
|
||||
]
|
||||
},
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
]
|
||||
@@ -228,7 +118,6 @@
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
@@ -236,22 +125,20 @@
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"A json file specifying where to overlay the thunks.",
|
||||
"This can be a filesystem path",
|
||||
"\teg: ~/MyThunkConfig.json",
|
||||
"Or this can be a named of a Thunk config file",
|
||||
"If the named config file exists in the FEX data folder folder the it will use that one",
|
||||
"\teg: $XDG_DATA_HOME/fex-emu/ThunkConfigs/<ThunkConfig name>",
|
||||
"If XDG_DATA_HOME is unset, ~/.local/share will be used in its place.",
|
||||
"\teg: $HOME/.local/share/fex-emu/ThunkConfigs/<ThunkConfig name>"
|
||||
"\teg: $HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>",
|
||||
"Or if you have XDG_DATA_HOME the config will search in that directory",
|
||||
"\teg: $XDG_DATA_HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>"
|
||||
]
|
||||
},
|
||||
"Env": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the emulated environment."
|
||||
]
|
||||
@@ -259,7 +146,6 @@
|
||||
"HostEnv": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the host environment.",
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
@@ -269,59 +155,15 @@
|
||||
"AdditionalArguments": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
},
|
||||
"DisableL2Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
]
|
||||
},
|
||||
"DynamicL1Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
]
|
||||
},
|
||||
"DynamicL1CacheIncreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "250",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should increase its size.",
|
||||
"Lower numbers means more aggressive scaling upward to the maximum size.",
|
||||
"Higher numbers means more conservative scaling, using less memory.",
|
||||
"Can potentially introduce stutters, more likely the higher the number.",
|
||||
"Don't have this number smaller than the decrease count!"
|
||||
]
|
||||
},
|
||||
"DynamicL1CacheDecreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "50",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should decrease its size.",
|
||||
"The higher the number, the more aggressively it reduces the L1 cache size.",
|
||||
"Lower numbers means more conservative memory savings.",
|
||||
"Can potentially introduce more stutters, more likely the higher the number.",
|
||||
"Don't have this number larger than the increase count!"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
"SingleStep": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Single stepping configuration."
|
||||
]
|
||||
@@ -329,7 +171,6 @@
|
||||
"GdbServer": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Enables the GDB server."
|
||||
]
|
||||
@@ -337,7 +178,6 @@
|
||||
"DumpIR": {
|
||||
"Type": "str",
|
||||
"Default": "no",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Folder to dump the IR in to.",
|
||||
"[no, stdout, stderr, server, <Folder>]"
|
||||
@@ -346,7 +186,6 @@
|
||||
"PassManagerDumpIR": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::PassManagerDumpIR::OFF",
|
||||
"AffectsCodeGen": "false",
|
||||
"Enums": {
|
||||
"BEFOREOPT": "beforeopt",
|
||||
"AFTEROPT": "afteropt",
|
||||
@@ -365,7 +204,6 @@
|
||||
"DumpGPRs": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"When the test harness ends, print the GPR state."
|
||||
]
|
||||
@@ -373,7 +211,6 @@
|
||||
"O0": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
@@ -381,7 +218,6 @@
|
||||
"GlobalJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name all JIT state as one symbol",
|
||||
"Useful for querying how much time is spent inside of the JIT",
|
||||
@@ -391,7 +227,6 @@
|
||||
"LibraryJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols grouped by library",
|
||||
"Useful for querying how much time is spent in each guest library",
|
||||
@@ -401,7 +236,6 @@
|
||||
"BlockJITNaming": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Uses JITSymbols to name JIT symbols",
|
||||
"Useful for determining hot blocks of code",
|
||||
@@ -411,7 +245,6 @@
|
||||
"GDBSymbols": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Integrates with GDB using the JIT interface.",
|
||||
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
|
||||
@@ -422,7 +255,6 @@
|
||||
"InjectLibSegFault": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Sets the environment variable LD_PRELOAD=libSegFault.so",
|
||||
"This allows the user to very easily enable libSegFault without dealing with environment variables",
|
||||
@@ -434,33 +266,22 @@
|
||||
"Disassemble": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::Disassemble::OFF",
|
||||
"AffectsCodeGen": "false",
|
||||
"Enums": {
|
||||
"DISPATCHER": "dispatcher",
|
||||
"BLOCKS": "blocks",
|
||||
"STATS": "stats"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the vixl disassembler for generated ARM code.",
|
||||
"Allows controlling of the vixl disassembler.",
|
||||
"\toff: No disassembly will be output",
|
||||
"\tdispatcher: Will enable disassembly of the JIT dispatcher loop",
|
||||
"\tblocks: Will enable disassembly of the translated instruction code blocks",
|
||||
"\tstats: Will print stats when disassembling the code"
|
||||
]
|
||||
},
|
||||
"X86Disassemble": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables x86/x86-64 guest disassembly output for compiled blocks.",
|
||||
"Requires FEX to be built with -DENABLE_ZYDIS=TRUE"
|
||||
]
|
||||
},
|
||||
"ForceSVEWidth": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Allows overriding the SVE width in the vixl simulator.",
|
||||
"Useful as a debugging feature."
|
||||
@@ -469,7 +290,6 @@
|
||||
"DisableTelemetry": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Disables telemetry at runtime.",
|
||||
"Useful for CI instcountCI mostly"
|
||||
@@ -480,7 +300,6 @@
|
||||
"SilentLog": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Disables logging"
|
||||
]
|
||||
@@ -488,36 +307,32 @@
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "server",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stderr, server, <Filename>]"
|
||||
"[stdout, stderr, server, <Filename>]"
|
||||
]
|
||||
},
|
||||
"TelemetryDirectory": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Redirects the telemetry folder that FEX usually writes to.",
|
||||
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/fex-emu/Telemetry/}"
|
||||
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
|
||||
]
|
||||
},
|
||||
"ProfileStats": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables FEX's low-overhead sampling profile statistics.",
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
]
|
||||
},
|
||||
"EnableGpuvisProfiling": {
|
||||
"TraceProfiler": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Enables profiling when FEX was built with the gpuvis profiler backend."
|
||||
"Enables FEX's trace profiler. Using gpuvis or tracy"
|
||||
]
|
||||
}
|
||||
},
|
||||
@@ -525,7 +340,6 @@
|
||||
"SMCChecks": {
|
||||
"Type": "uint8",
|
||||
"Default": "FEXCore::Config::CONFIG_SMC_MTRACK",
|
||||
"AffectsCodeGen": "true",
|
||||
"TextDefault": "mtrack",
|
||||
"ArgumentHandler": "SMCCheckHandler",
|
||||
"Desc": [
|
||||
@@ -538,7 +352,6 @@
|
||||
"TSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Controls TSO IR ops.",
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
@@ -547,7 +360,6 @@
|
||||
"VectorTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
|
||||
]
|
||||
@@ -555,7 +367,6 @@
|
||||
"MemcpySetTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
|
||||
"Only affects REP MOVS and REP STOS instructions"
|
||||
@@ -564,7 +375,6 @@
|
||||
"HalfBarrierTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if unaligned loads and stores should be backpatched to half-barrier atomics.",
|
||||
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
|
||||
@@ -573,41 +383,46 @@
|
||||
"StrictInProcessSplitLocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Strict global lock when handling an unaligned atomic that crosses a 16-byte or cacheline granularity",
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
]
|
||||
},
|
||||
"KernelUnalignedAtomicBackpatching": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"When the kernel unaligned atomic handler is enabled, use backpatching to reduce kernel context switches."
|
||||
]
|
||||
},
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Use volatile metadata in PE files to inform TSO instructions when available.",
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When enabled enables an optimization around flags.",
|
||||
"Assumes flags are not used across cals.",
|
||||
"Hand-written assembly can violate this assumption."
|
||||
]
|
||||
},
|
||||
"ParanoidTSO": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Makes TSO operations even more strict.",
|
||||
"Forces vector loadstores to also become atomic."
|
||||
]
|
||||
},
|
||||
"StallProcess": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Forces a process to stall out on initialization",
|
||||
"Useful for a process that keeps restarting and doesn't work"
|
||||
@@ -616,7 +431,6 @@
|
||||
"HideHypervisorBit": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Hides the hypervisor CPUID bit when set.",
|
||||
"Should only be used for applications that have issues with this set."
|
||||
@@ -625,7 +439,6 @@
|
||||
"StartupSleep": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Sleeps the process at startup for a duration of seconds.",
|
||||
"Useful if an application crashes too quickly to attach a debugger."
|
||||
@@ -634,7 +447,6 @@
|
||||
"StartupSleepProcName": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Contrains the startup sleep to only apply to processes that match this name."
|
||||
]
|
||||
@@ -642,7 +454,6 @@
|
||||
"MonoHacks": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Permits a hook-based SMC approach and smaller JIT blocks when mono is detected."
|
||||
]
|
||||
@@ -652,7 +463,6 @@
|
||||
"ServerSocketPath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Override for a FEXServer socket path. Only useful for chroots."
|
||||
]
|
||||
@@ -660,7 +470,6 @@
|
||||
"NeedsSeccomp": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Disables inline syscalls in order to support seccomp handling"
|
||||
]
|
||||
@@ -668,7 +477,6 @@
|
||||
"ExtendedVolatileMetadata": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Configuration provided volatile metadata. Only implemented for WoW64/arm64ec.",
|
||||
"Limited in its use but can be handy.",
|
||||
@@ -693,18 +501,15 @@
|
||||
"Misc": {
|
||||
"INTERPRETER_INSTALLED": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false"
|
||||
"Default": "false"
|
||||
},
|
||||
"APP_FILENAME": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false"
|
||||
"Default": ""
|
||||
},
|
||||
"APP_CONFIG_NAME": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"This is the application config name that has been loaded.",
|
||||
"This differs from APP_FILENAME in two ways",
|
||||
@@ -715,29 +520,16 @@
|
||||
},
|
||||
"IS64BIT_MODE": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "false",
|
||||
"Comment": "Technically affects codegen, but this is serialized elsewhere."
|
||||
"Default": "false"
|
||||
},
|
||||
"DISABLE_VIXL_INDIRECT_RUNTIME_CALLS": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Comment": "Technically affects codegen, but only shows up in the test harness.",
|
||||
"Desc": [
|
||||
"This option is used for the InstructionCountCI so it can generate the same codegen between Arm64 hosts and vixl simulator hosts.",
|
||||
"Vixl simulator indirect runtime calls are a special hlt instruction with metadata after it. Effectively making a custom call instruction.",
|
||||
"With visual simulator calls disabled, the code generation would be the same as on a native Arm64 host, but running the code is broken."
|
||||
]
|
||||
},
|
||||
"CONFIG_VERSION": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "true",
|
||||
"Comment": [
|
||||
"Meta option that if config has ever changed definitions dramatically enough that we can rev the version.",
|
||||
"Be mindful that this will invalidate all caches!"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -53,12 +53,6 @@ FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionN
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const {
|
||||
return Thread->CPUBackend->IsAddressInCodeBuffer(Address) || CodeCache.IsAddressInMappedCodeBuffer(Address);
|
||||
return Thread->CPUBackend->IsAddressInCodeBuffer(Address);
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::RequiresRelocatableConstants() const {
|
||||
// Support relocation when generating a cache or when generating reference code for validation
|
||||
return CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHEVALIDATION() || DiskCache.IsWritingDiskCache();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Context
|
||||
@@ -4,13 +4,12 @@
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/SharedCodeBufferManager.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include <Interface/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/DiskCache.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
@@ -30,7 +29,6 @@
|
||||
namespace FEXCore {
|
||||
class SignalDelegator;
|
||||
class ThunkHandler;
|
||||
struct LookupCacheWriteLockToken;
|
||||
|
||||
namespace Core {
|
||||
struct DebugData;
|
||||
@@ -63,86 +61,32 @@ struct CustomIRResult {
|
||||
, Data(Data) {}
|
||||
};
|
||||
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
constexpr static bool BLOCK_DEBUGGING = false;
|
||||
|
||||
class CodeCache : public AbstractCodeCache {
|
||||
public:
|
||||
CodeCache(ContextImpl&);
|
||||
~CodeCache();
|
||||
|
||||
ContextImpl& CTX;
|
||||
fextl::unique_ptr<ContextImpl> ValidationCTX;
|
||||
fextl::unique_ptr<Core::InternalThreadState> ValidationThread;
|
||||
FEXCore::Core::CPUState::gdt_segment ValidationGDT[32] {};
|
||||
bool IsGeneratingCache = false;
|
||||
|
||||
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
|
||||
FEX_CONFIG_OPT(EnableLazyCodeCaching, ENABLELAZYCODECACHINGWIP);
|
||||
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
|
||||
|
||||
uint64_t ComputeCodeMapId(std::string_view Filename, int FD) override;
|
||||
void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
|
||||
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
|
||||
|
||||
fextl::unique_ptr<MappedCodeCacheFile> LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo&, uint64_t FileStartVA) override;
|
||||
|
||||
bool EnableLoadedSection(Core::InternalThreadState*, MappedCodeCacheFile&, const ExecutableFileSectionInfo&) override;
|
||||
|
||||
void FinalizeCodePages(MappedCodeCacheFile&, std::span<std::byte> CodeRange) override;
|
||||
|
||||
/**
|
||||
* Performs expensive extra validation on the loaded code cache data.
|
||||
*
|
||||
* This kicks off an in-process recompile of all cached blocks and compares
|
||||
* them with the cached data. Differences will be reported as fatal errors,
|
||||
* which can uncover bugs like for example:
|
||||
* - mismatches of the JIT configuration used during cache generation
|
||||
* - hidden position dependencies due to missing FEX relocations
|
||||
* - incorrect instruction padding
|
||||
*/
|
||||
void Validate(const ExecutableFileSectionInfo&, fextl::set<uint64_t> GuestBlocks, const fextl::set<uint64_t>& HostBlocks,
|
||||
std::span<std::byte> CachedCode);
|
||||
|
||||
void InitiateCacheGeneration() override {
|
||||
IsGeneratingCache = true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies a set of FEX relocations to the given code section.
|
||||
*
|
||||
* FEX relocations describe runtime-dependencies of FEX-generated code.
|
||||
* When loading a code cache, they are used to move cached code to the
|
||||
* dynamically chosen base address of the guest binary.
|
||||
*
|
||||
* Conversely, relocations are applied in reverse when writing code caches
|
||||
* to ensure consistency across generation runs.
|
||||
*
|
||||
* Note that FEX relocations are unrelated to ELF/PE relocations.
|
||||
*
|
||||
* @param GuestDelta Guest address offset to apply to RIP-relative data
|
||||
* @param ForStorage True for serializing data (producing deterministic output); false for de-serializing it (resolving dynamic symbols)
|
||||
*
|
||||
* @return Returns true on success
|
||||
*/
|
||||
[[nodiscard]]
|
||||
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations, bool ForStorage);
|
||||
|
||||
// Same but on disk cache packed relocations
|
||||
[[nodiscard]]
|
||||
bool ApplyPackedCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const DiskCache::BlobSmallRelocation> SmallRelocs,
|
||||
std::span<const DiskCache::BlobThunkRelocation> ThunkRelocs);
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::SharedCodeBufferManager {
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
|
||||
void ExecuteThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
bool CheckIfBlockIsCacheable(FEXCore::Core::InternalThreadState&, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
@@ -160,32 +104,32 @@ public:
|
||||
void SetXMMRegistersFromState(FEXCore::Core::InternalThreadState* Thread, const __uint128_t* XMM_Low, const __uint128_t* YMM_High) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread.
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread();
|
||||
* - Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
* - Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP] = InitialStack;
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - CTX->ExecuteThread(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(NewState);
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
* - ThreadHandler calls `CTX->ExecuteThread(Thread)`
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(CopyOfThreadState);
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - ExecuteThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(NewState);
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(const FEXCore::Core::CPUState* NewThreadState) override;
|
||||
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState) override;
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
@@ -206,27 +150,15 @@ public:
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
virtual void InitDiskCache() override {}
|
||||
|
||||
CodeCache& GetCodeCache() override {
|
||||
return CodeCache;
|
||||
}
|
||||
|
||||
void SetCodeMapWriter(fextl::unique_ptr<CodeMapWriter> Writer) override {
|
||||
CodeMapWriter = std::move(Writer);
|
||||
}
|
||||
|
||||
void FlushAndCloseCodeMap() override {
|
||||
if (CodeMapWriter) {
|
||||
CodeMapWriter.reset();
|
||||
}
|
||||
}
|
||||
|
||||
void OnCodeBufferAllocated(const std::shared_ptr<CPU::CodeBuffer>&) override;
|
||||
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
|
||||
void InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex& GetCodeInvalidationMutex() override {
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start,
|
||||
uint64_t Length) override;
|
||||
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
|
||||
@@ -249,100 +181,7 @@ public:
|
||||
}
|
||||
|
||||
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
|
||||
std::atomic<uint64_t>& GetMonoBackPatcherBlock() {
|
||||
return MonoBackpatcherBlock;
|
||||
}
|
||||
|
||||
// Manual debugging tooling which is useful for developers.
|
||||
struct TrackingEmpty {
|
||||
// RIP stepping handling
|
||||
virtual void AddSingleStepTarget(uint64_t GuestRIP) {}
|
||||
virtual void AddSingleStepTargetRange(uint64_t RIPBegin, uint64_t RipEnd) {}
|
||||
virtual void AllTargetSingleStep() {}
|
||||
virtual void RemoveSingleStepTarget(uint64_t GuestRIP) {}
|
||||
virtual bool IsSingleStepTarget(uint64_t GuestRIP) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Watchpoints
|
||||
virtual void AddWriteWatchPoint(uint64_t Ptr) {}
|
||||
virtual void AddReadWatchPoint(uint64_t Ptr) {}
|
||||
virtual bool ContainsWriteWatchPoint(uint64_t Ptr, size_t Size) {
|
||||
return false;
|
||||
}
|
||||
virtual bool ContainsReadWatchPoint(uint64_t Ptr, size_t Size) {
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
struct TrackingPossible final : public TrackingEmpty {
|
||||
void AddSingleStepTarget(uint64_t GuestRIP) override {
|
||||
SingleStepTargets.emplace(GuestRIP);
|
||||
}
|
||||
|
||||
virtual void AddSingleStepTargetRange(uint64_t RIPBegin, uint64_t RIPEnd) override {
|
||||
SingleStepRanges.emplace_back(Range {RIPBegin, RIPEnd});
|
||||
}
|
||||
|
||||
void RemoveSingleStepTarget(uint64_t GuestRIP) override {
|
||||
SingleStepTargets.erase(GuestRIP);
|
||||
}
|
||||
|
||||
void AllTargetSingleStep() override {
|
||||
SingleStepEverything = true;
|
||||
}
|
||||
|
||||
bool IsSingleStepTarget(uint64_t GuestRIP) override {
|
||||
return SingleStepEverything || SingleStepTargets.contains(GuestRIP) || IsInRange(GuestRIP);
|
||||
}
|
||||
|
||||
void AddWriteWatchPoint(uint64_t Ptr) override {
|
||||
WatchWriteTargets.emplace(Ptr);
|
||||
}
|
||||
|
||||
void AddReadWatchPoint(uint64_t Ptr) override {
|
||||
WatchReadTargets.emplace(Ptr);
|
||||
}
|
||||
|
||||
bool ContainsWriteWatchPoint(uint64_t Ptr, size_t Size) override {
|
||||
return ContainsRange(WatchWriteTargets, Ptr, Size);
|
||||
}
|
||||
|
||||
bool ContainsReadWatchPoint(uint64_t Ptr, size_t Size) override {
|
||||
return ContainsRange(WatchReadTargets, Ptr, Size);
|
||||
}
|
||||
|
||||
private:
|
||||
bool SingleStepEverything {};
|
||||
fextl::set<uint64_t> SingleStepTargets {};
|
||||
fextl::set<uint64_t> WatchWriteTargets {};
|
||||
fextl::set<uint64_t> WatchReadTargets {};
|
||||
struct Range {
|
||||
uint64_t Begin, End;
|
||||
};
|
||||
fextl::vector<Range> SingleStepRanges {};
|
||||
|
||||
bool IsInRange(uint64_t RIP) const {
|
||||
return std::ranges::any_of(SingleStepRanges, [RIP](const auto& range) { return RIP >= range.Begin && RIP <= range.End; });
|
||||
}
|
||||
|
||||
static bool ContainsRange(const fextl::set<uint64_t>& Set, uint64_t Ptr, size_t Size) {
|
||||
for (auto it = Set.lower_bound(Ptr); it != Set.end(); --it) {
|
||||
auto Watch = *it;
|
||||
if (Watch < Ptr) {
|
||||
break;
|
||||
}
|
||||
if (Watch >= Ptr && Watch < (Ptr + Size)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
};
|
||||
using TrackingStructure = std::conditional<BLOCK_DEBUGGING, TrackingPossible, TrackingEmpty>::type;
|
||||
|
||||
TrackingStructure BlockDebuggerTracker {};
|
||||
public:
|
||||
struct {
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
@@ -358,6 +197,7 @@ public:
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
@@ -365,6 +205,7 @@ public:
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
@@ -373,7 +214,7 @@ public:
|
||||
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
|
||||
} Config;
|
||||
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex CodeInvalidationMutex {};
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
uint32_t StrictSplitLockMutex {};
|
||||
|
||||
@@ -384,14 +225,15 @@ public:
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
DiskCache::DiskCache DiskCache;
|
||||
CodeCache CodeCache;
|
||||
fextl::unique_ptr<CodeMapWriter> CodeMapWriter;
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
ContextImpl(const FEXCore::HostFeatures& Features);
|
||||
|
||||
static bool ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
|
||||
|
||||
// This is used as a replacement for the SMC writes in the mono callsite backpatcher that avoids atomic operations
|
||||
@@ -426,9 +268,9 @@ public:
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator {"FEXMem_OpDispatcher"};
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator {"FEXMem_Frontend"};
|
||||
FEXCore::Utils::PooledAllocatorVirtualWithGuard CPUBackendAllocator {"FEXMem_CPUBackend"};
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual CPUBackendAllocator;
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const {
|
||||
@@ -462,8 +304,6 @@ public:
|
||||
return Config.MonoHacks && MonoDetected;
|
||||
}
|
||||
|
||||
bool RequiresRelocatableConstants() const;
|
||||
|
||||
protected:
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
@@ -471,6 +311,10 @@ protected:
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
VectorAtomicTSOEmulationEnabled = false;
|
||||
MemcpyAtomicTSOEmulationEnabled = false;
|
||||
} else if (Config.ParanoidTSO) {
|
||||
AtomicTSOEmulationEnabled = true;
|
||||
VectorAtomicTSOEmulationEnabled = true;
|
||||
MemcpyAtomicTSOEmulationEnabled = true;
|
||||
} else {
|
||||
AtomicTSOEmulationEnabled = Config.TSOEnabled;
|
||||
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
@@ -509,8 +353,5 @@ private:
|
||||
|
||||
bool MonoDetected = false;
|
||||
std::atomic<uint64_t> MonoBackpatcherBlock;
|
||||
|
||||
std::mutex CodeBufferListLock;
|
||||
fextl::vector<std::weak_ptr<CPU::CodeBuffer>> CodeBufferList;
|
||||
};
|
||||
} // namespace FEXCore::Context
|
||||
@@ -7,7 +7,7 @@
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
@@ -51,8 +51,8 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
|
||||
return Tmp ?: IREmit->Constant(0);
|
||||
}
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
|
||||
bool Vector, IR::OpSize AccessSize) {
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize) {
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
@@ -103,7 +103,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
@@ -111,17 +111,15 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
|
||||
if (AtomicTSO) {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
AddressMode B = A;
|
||||
|
||||
// ScaledRegisterLoadstore
|
||||
if (B.Index && B.Segment) {
|
||||
B.Base = IREmit->Add(GPRSize, B.Base, B.Segment);
|
||||
} else if (B.Segment) {
|
||||
B.Index = B.Segment;
|
||||
B.IndexScale = 1;
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = IREmit->Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
|
||||
return B;
|
||||
return A;
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
@@ -136,7 +134,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -11,18 +11,17 @@ struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
int64_t Offset = 0;
|
||||
|
||||
MemOffsetType IndexType = MemOffsetType::SXTX;
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
|
||||
bool Vector, IR::OpSize AccessSize);
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize);
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
}; // namespace FEXCore::IR
|
||||
@@ -1,10 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
@@ -41,7 +41,7 @@ namespace FEXCore::CPU {
|
||||
// r19-r29 and SP.
|
||||
|
||||
namespace x64 {
|
||||
#ifndef ARCHITECTURE_arm64ec
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
// Note that rax/rdx are rearranged here so we can coalesce cmpxchg.
|
||||
constexpr std::array<ARMEmitter::Register, 18> SRA = {
|
||||
@@ -360,7 +360,6 @@ namespace x32 {
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr, size_t size)
|
||||
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
|
||||
, EmitterCTX {ctx}
|
||||
, SupportCodeRelocations {ctx->RequiresRelocatableConstants()}
|
||||
#ifdef VIXL_SIMULATOR
|
||||
, Simulator {&SimDecoder, stdout, vixl::aarch64::SimStack(SimulatorStackSize).Allocate()}
|
||||
#endif
|
||||
@@ -418,54 +417,36 @@ FEXCore::X86State::X86Reg Arm64Emitter::GetX86RegRelationToARMReg(ARMEmitter::Re
|
||||
return FEXCore::X86State::X86Reg::REG_INVALID;
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad, int MaxBytes) {
|
||||
bool NOPPad = false;
|
||||
if (Pad == PadType::DOPAD) {
|
||||
NOPPad = true;
|
||||
} else if (Pad == PadType::NOPAD) {
|
||||
NOPPad = false;
|
||||
} else if (Pad == PadType::AUTOPAD) {
|
||||
// Force NOP padding to ensure relocated constants always have enough encoding space available
|
||||
NOPPad = SupportCodeRelocations;
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
const auto UpperBound = Is64Bit ? 4 : 2;
|
||||
int Segments = MaxBytes ? (MaxBytes / 2) : UpperBound;
|
||||
|
||||
LOGMAN_THROW_A_FMT(MaxBytes >= 0 && MaxBytes <= (UpperBound * 2) && (MaxBytes & 1) == 0,
|
||||
"MaxBytes must be bounded in the range of [0, {}] and 16-bit aligned", UpperBound);
|
||||
// If MaxBytes specified then make sure to sanity check incoming data.
|
||||
LOGMAN_THROW_A_FMT(MaxBytes == 0 || (Constant >> (MaxBytes * 8)) == 0, "MaxBytes provided but data can't fit within provided range.");
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
|
||||
if (Is64Bit && ((~Constant) >> 16) == 0) {
|
||||
movn(s, Reg, (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop();
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
|
||||
movn(s, Reg, (~Constant) & 0xFFFF);
|
||||
return;
|
||||
}
|
||||
|
||||
if ((Constant >> 32) == 0 && !NOPPad) {
|
||||
if ((Constant >> 32) == 0) {
|
||||
// If the upper 32-bits is all zero, we can now switch to a 32-bit move.
|
||||
// NOTE: The NOP padding code does not appropriately adjust to this yet,
|
||||
// so we skip this optimization in that case
|
||||
s = ARMEmitter::Size::i32Bit;
|
||||
Is64Bit = false;
|
||||
Segments = std::min(Segments, 2);
|
||||
Segments = 2;
|
||||
}
|
||||
|
||||
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
|
||||
movn(s, Reg.W(), (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop();
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
|
||||
movn(s, Reg.W(), (~Constant) & 0xFFFF);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -486,24 +467,24 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
// `movz` is better than `orr` since hardware will rename or merge if possible when `movz` is used.
|
||||
const auto IsImm = ARMEmitter::Emitter::IsImmLogical(Constant, RegSizeInBits(s));
|
||||
if (IsImm) {
|
||||
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
|
||||
if (NOPPad) {
|
||||
nop();
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// If we can't handle negatives with the orr, try with movn+movk
|
||||
if (Is64Bit && ((~Constant) >> 32) == 0) {
|
||||
movn(s, Reg, (~Constant) & 0xFFFF);
|
||||
movk(s, Reg, (Constant >> 16) & 0xFFFF, 16);
|
||||
if (NOPPad) {
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
movn(s, Reg, (~Constant) & 0xFFFF);
|
||||
movk(s, Reg, (Constant >> 16) & 0xFFFF, 16);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -587,8 +568,8 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
|
||||
}};
|
||||
|
||||
for (const auto& [rt, rt2] : CalleeSaved) {
|
||||
stp<ARMEmitter::IndexType::PRE>(rt, rt2, ARMEmitter::Reg::rsp, -16);
|
||||
for (auto& RegPair : CalleeSaved) {
|
||||
stp<ARMEmitter::IndexType::PRE>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, -16);
|
||||
}
|
||||
|
||||
// Additionally we need to store the lower 64bits of v8-v15
|
||||
@@ -605,8 +586,9 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We just saved x19 so it is safe
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r19, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
for (const auto& [rt, rt2, rt3, rt4] : FPRs) {
|
||||
st4(ARMEmitter::SubRegSize::i64Bit, rt, rt2, rt3, rt4, 0, ARMEmitter::Reg::r19, 32);
|
||||
for (auto& RegQuad : FPRs) {
|
||||
st4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
|
||||
ARMEmitter::Reg::r19, 32);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -616,8 +598,9 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
}};
|
||||
|
||||
for (const auto& [rt, rt2, rt3, rt4] : FPRs) {
|
||||
ld4(ARMEmitter::SubRegSize::i64Bit, rt, rt2, rt3, rt4, 0, ARMEmitter::Reg::rsp, 32);
|
||||
for (auto& RegQuad : FPRs) {
|
||||
ld4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
|
||||
ARMEmitter::Reg::rsp, 32);
|
||||
}
|
||||
|
||||
constexpr static std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
|
||||
@@ -629,12 +612,12 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
|
||||
}};
|
||||
|
||||
for (const auto& [rt, rt2] : CalleeSaved) {
|
||||
ldp<ARMEmitter::IndexType::POST>(rt, rt2, ARMEmitter::Reg::rsp, 16);
|
||||
for (auto& RegPair : CalleeSaved) {
|
||||
ldp<ARMEmitter::IndexType::POST>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, const FillSpecialRegsOptions& Options) {
|
||||
void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Enable AFP features when filling JIT state.
|
||||
@@ -650,7 +633,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
|
||||
if (Options.SetFIZ) {
|
||||
if (SetFIZ) {
|
||||
// Insert MXCSR.DAZ in to FIZ
|
||||
ldr(TmpReg2.W(), STATE.R(), offsetof(FEXCore::Core::CPUState, mxcsr));
|
||||
bfxil(ARMEmitter::Size::i64Bit, TmpReg, TmpReg2, 6, 1);
|
||||
@@ -660,7 +643,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
}
|
||||
#endif
|
||||
|
||||
if (Options.SetPredRegs && EmitterCTX->HostFeatures.SupportsSVE()) {
|
||||
if (SetPredRegs && (EmitterCTX->HostFeatures.SupportsSVE256 || EmitterCTX->HostFeatures.SupportsSVE128)) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
@@ -678,7 +661,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, SpillStaticRegOptions Options) {
|
||||
void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Disable AFP features when spilling registers.
|
||||
@@ -699,37 +682,35 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, SpillStaticRegOp
|
||||
}
|
||||
#endif
|
||||
|
||||
if (Options.NZCV) {
|
||||
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
|
||||
// is always static and almost certainly clobbered by the subsequent code.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
}
|
||||
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
|
||||
// is always static and almost certainly clobbered by the subsequent code.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
unsigned PFAFSpillMask = Options.GPRSpillMask & PFAFMask;
|
||||
Options.GPRSpillMask &= ~PFAFSpillMask;
|
||||
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
|
||||
GPRSpillMask &= ~PFAFSpillMask;
|
||||
|
||||
str(REG_CALLRET_SP, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.callret_sp));
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i + 1];
|
||||
if (((1U << Reg1.Idx()) & Options.GPRSpillMask) && ((1U << Reg2.Idx()) & Options.GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i));
|
||||
} else if (((1U << Reg1.Idx()) & Options.GPRSpillMask)) {
|
||||
str(Reg1.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i));
|
||||
} else if (((1U << Reg2.Idx()) & Options.GPRSpillMask)) {
|
||||
str(Reg2.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i + 1));
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) && ((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
} else if (((1U << Reg1.Idx()) & GPRSpillMask)) {
|
||||
str(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
} else if (((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
str(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i + 1]));
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (Options.NZCV && PFAFSpillMask) {
|
||||
if (PFAFSpillMask) {
|
||||
auto PFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
@@ -738,21 +719,21 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, SpillStaticRegOp
|
||||
stp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), PFOffset);
|
||||
}
|
||||
|
||||
if (Options.FPRs) {
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
|
||||
if (((1U << Reg.Idx()) & Options.FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TmpReg, ARRAY_OFFSETOF(Core::CpuStateFrame, State.xmm.avx.data, i));
|
||||
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TmpReg, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B, STATE.R(), TmpReg);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (Options.GPRSpillMask && Options.FPRSpillMask == ~0U) {
|
||||
if (GPRSpillMask && FPRSpillMask == ~0U) {
|
||||
// Optimize the common case where we can spill four registers per instruction
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data));
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
@@ -765,12 +746,12 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, SpillStaticRegOp
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & Options.FPRSpillMask) && ((1U << Reg2.Idx()) & Options.FPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i));
|
||||
} else if (((1U << Reg1.Idx()) & Options.FPRSpillMask)) {
|
||||
str(Reg1.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i));
|
||||
} else if (((1U << Reg2.Idx()) & Options.FPRSpillMask)) {
|
||||
str(Reg2.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i + 1));
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) && ((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
} else if (((1U << Reg1.Idx()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
} else if (((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i + 1][0]));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -778,7 +759,8 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, SpillStaticRegOp
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(FillStaticRegOptions Options) {
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask, std::optional<ARMEmitter::Register> OptionalReg,
|
||||
std::optional<ARMEmitter::Register> OptionalReg2) {
|
||||
auto FindTempReg = [this](uint32_t* GPRFillMask) -> std::optional<ARMEmitter::Register> {
|
||||
for (auto Reg : StaticRegisters) {
|
||||
if (((1U << Reg.Idx()) & *GPRFillMask)) {
|
||||
@@ -789,23 +771,22 @@ void Arm64Emitter::FillStaticRegs(FillStaticRegOptions Options) {
|
||||
return std::nullopt;
|
||||
};
|
||||
|
||||
LOGMAN_THROW_A_FMT(Options.GPRFillMask != 0, "Must fill at least 2 GPRs for a temp");
|
||||
uint32_t TempGPRFillMask = Options.GPRFillMask;
|
||||
if (!Options.OptionalReg.has_value()) {
|
||||
Options.OptionalReg = FindTempReg(&TempGPRFillMask);
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 2 GPRs for a temp");
|
||||
uint32_t TempGPRFillMask = GPRFillMask;
|
||||
if (!OptionalReg.has_value()) {
|
||||
OptionalReg = FindTempReg(&TempGPRFillMask);
|
||||
}
|
||||
|
||||
if (!Options.OptionalReg2.has_value()) {
|
||||
Options.OptionalReg2 = FindTempReg(&TempGPRFillMask);
|
||||
if (!OptionalReg2.has_value()) {
|
||||
OptionalReg2 = FindTempReg(&TempGPRFillMask);
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(Options.OptionalReg.has_value() && Options.OptionalReg2.has_value(), "Didn't have an SRA register to use as a "
|
||||
"temporary while "
|
||||
"spilling!");
|
||||
LOGMAN_THROW_A_FMT(OptionalReg.has_value() && OptionalReg2.has_value(), "Didn't have an SRA register to use as a temporary while "
|
||||
"spilling!");
|
||||
|
||||
auto TmpReg = *Options.OptionalReg;
|
||||
auto TmpReg2 = *Options.OptionalReg2;
|
||||
auto TmpReg = *OptionalReg;
|
||||
auto TmpReg2 = *OptionalReg2;
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _M_ARM_64EC
|
||||
// Load STATE in from the CPU area as x28 is not callee saved in the ARM64EC ABI.
|
||||
ldr(TmpReg.X(), ARMEmitter::Reg::r18, TEB_CPU_AREA_OFFSET);
|
||||
ldr(STATE, TmpReg, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
@@ -813,33 +794,31 @@ void Arm64Emitter::FillStaticRegs(FillStaticRegOptions Options) {
|
||||
|
||||
ldr(REG_CALLRET_SP, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.callret_sp));
|
||||
|
||||
if (Options.NZCV) {
|
||||
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
|
||||
// is always static and was almost certainly clobbered.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
|
||||
}
|
||||
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
|
||||
// is always static and was almost certainly clobbered.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
|
||||
|
||||
FillSpecialRegs(TmpReg, TmpReg2, {.SetFIZ = true, .SetPredRegs = Options.FPRs});
|
||||
FillSpecialRegs(TmpReg, TmpReg2, true, FPRs);
|
||||
|
||||
if (Options.FPRs) {
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
if (((1U << Reg.Idx()) & Options.FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TmpReg, ARRAY_OFFSETOF(Core::CpuStateFrame, State.xmm.avx.data, i));
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TmpReg, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B.Zeroing(), STATE.R(), TmpReg);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (Options.GPRFillMask && Options.FPRFillMask == ~0U) {
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data));
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
@@ -852,12 +831,12 @@ void Arm64Emitter::FillStaticRegs(FillStaticRegOptions Options) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & Options.FPRFillMask) && ((1U << Reg2.Idx()) & Options.FPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i));
|
||||
} else if (((1U << Reg1.Idx()) & Options.FPRFillMask)) {
|
||||
ldr(Reg1.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i));
|
||||
} else if (((1U << Reg2.Idx()) & Options.FPRFillMask)) {
|
||||
ldr(Reg2.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i + 1));
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) && ((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
} else if (((1U << Reg1.Idx()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
} else if (((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i + 1][0]));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -866,23 +845,23 @@ void Arm64Emitter::FillStaticRegs(FillStaticRegOptions Options) {
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
uint32_t PFAFFillMask = Options.GPRFillMask & PFAFMask;
|
||||
Options.GPRFillMask &= ~PFAFMask;
|
||||
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
|
||||
GPRFillMask &= ~PFAFMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i + 1];
|
||||
if (((1U << Reg1.Idx()) & Options.GPRFillMask) && ((1U << Reg2.Idx()) & Options.GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i));
|
||||
} else if ((1U << Reg1.Idx()) & Options.GPRFillMask) {
|
||||
ldr(Reg1.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i));
|
||||
} else if ((1U << Reg2.Idx()) & Options.GPRFillMask) {
|
||||
ldr(Reg2.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i + 1));
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) && ((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
} else if ((1U << Reg1.Idx()) & GPRFillMask) {
|
||||
ldr(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
} else if ((1U << Reg2.Idx()) & GPRFillMask) {
|
||||
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i + 1]));
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (Options.NZCV && PFAFFillMask) {
|
||||
if (PFAFFillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
|
||||
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
@@ -1057,11 +1036,7 @@ size_t Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, boo
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
// Spill the static registers.
|
||||
SpillStaticRegs(TmpReg, {
|
||||
.GPRSpillMask = PreserveSRAMask,
|
||||
.FPRSpillMask = PreserveSRAFPRMask,
|
||||
.FPRs = FPRs,
|
||||
});
|
||||
SpillStaticRegs(TmpReg, true, PreserveSRAMask, PreserveSRAFPRMask);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
@@ -1108,11 +1083,7 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
}
|
||||
|
||||
// Fill the static registers.
|
||||
FillStaticRegs({
|
||||
.GPRFillMask = PreserveSRAMask,
|
||||
.FPRFillMask = PreserveSRAFPRMask,
|
||||
.FPRs = FPRs,
|
||||
});
|
||||
FillStaticRegs(FPRs, PreserveSRAMask, PreserveSRAFPRMask);
|
||||
|
||||
// Pop the vector registers.
|
||||
PopVectorRegisters(CanUseSVE256, DynamicFPRs);
|
||||
@@ -1123,7 +1094,6 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
|
||||
void Arm64Emitter::Align16B() {
|
||||
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
|
||||
LOGMAN_THROW_A_FMT((CurrentOffset & 3) == 0, "Can't Align16B code that isn't 4-byte aligned!");
|
||||
for (uint64_t i = (-CurrentOffset & 0xF); i != 0; i -= 4) {
|
||||
nop();
|
||||
}
|
||||
|
||||
@@ -1,38 +1,36 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#endif
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <optional>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEXCore::X86State {
|
||||
enum X86Reg : uint32_t;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Contains the address to the currently available CPU state
|
||||
constexpr auto STATE = ARMEmitter::XReg::x28;
|
||||
|
||||
#ifndef ARCHITECTURE_arm64ec
|
||||
#ifndef _M_ARM_64EC
|
||||
// GPR temporaries. Only x3 can be used across spill boundaries
|
||||
// so if these ever need to change, be very careful about that.
|
||||
constexpr auto TMP1 = ARMEmitter::XReg::x0;
|
||||
@@ -106,20 +104,9 @@ constexpr ARMEmitter::PRegister PRED_TMP_32B = ARMEmitter::PReg::p7;
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public ARMEmitter::Emitter {
|
||||
public:
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
|
||||
enum class PadType {
|
||||
// Explicitly does not need padding, even if code-caching is enabled.
|
||||
NOPAD,
|
||||
// Explicitly needs padding, even if code-caching is disabled.
|
||||
DOPAD,
|
||||
// Choose to pad or not depending on if code-caching is enabled.
|
||||
AUTOPAD,
|
||||
};
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad = PadType::NOPAD, int MaxBytes = 0);
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
@@ -129,55 +116,18 @@ protected:
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
|
||||
bool SupportCodeRelocations;
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
struct FillSpecialRegsOptions {
|
||||
// Whether or not to set the FPCR.FIZ (flush inputs to zero) bit in the FPCR to
|
||||
// the current value of the emulated MXCSR.DAZ bit.
|
||||
// Will only attempt to do so, even when set to true, if and only if the host system
|
||||
// supports FEAT_AFP.
|
||||
bool SetFIZ {};
|
||||
|
||||
// Whether or not FillSpecialRegs should load our SVE predicate temporaries
|
||||
// with certain canned values that accelerate some operations. Will (obviously)
|
||||
// not load predicates, even if set to true, on host systems that do not support SVE.
|
||||
bool SetPredRegs {};
|
||||
};
|
||||
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, const FillSpecialRegsOptions& Options);
|
||||
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
|
||||
|
||||
// Correlate an ARM register back to an x86 register index.
|
||||
// Returning REG_INVALID if there was no mapping.
|
||||
FEXCore::X86State::X86Reg GetX86RegRelationToARMReg(ARMEmitter::Register Reg);
|
||||
|
||||
struct SpillStaticRegOptions final {
|
||||
uint32_t GPRSpillMask {~0U};
|
||||
uint32_t FPRSpillMask {~0U};
|
||||
bool FPRs {true};
|
||||
bool NZCV {true};
|
||||
};
|
||||
|
||||
struct FillStaticRegOptions final {
|
||||
std::optional<ARMEmitter::Register> OptionalReg {std::nullopt};
|
||||
std::optional<ARMEmitter::Register> OptionalReg2 {std::nullopt};
|
||||
uint32_t GPRFillMask {~0U};
|
||||
uint32_t FPRFillMask {~0U};
|
||||
bool FPRs {true};
|
||||
bool NZCV {true};
|
||||
};
|
||||
|
||||
void SpillStaticRegs(ARMEmitter::Register TmpReg, SpillStaticRegOptions Options);
|
||||
void FillStaticRegs(FillStaticRegOptions Options);
|
||||
|
||||
|
||||
void SpillStaticRegs(ARMEmitter::Register TmpReg) {
|
||||
// Work around a clang bug: https://bugs.llvm.org/show_bug.cgi?id=36684
|
||||
SpillStaticRegs(TmpReg, {});
|
||||
}
|
||||
|
||||
void FillStaticRegs() {
|
||||
// Work around a clang bug: https://bugs.llvm.org/show_bug.cgi?id=36684
|
||||
FillStaticRegs({});
|
||||
}
|
||||
void SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U,
|
||||
std::optional<ARMEmitter::Register> OptionalReg = std::nullopt,
|
||||
std::optional<ARMEmitter::Register> OptionalReg2 = std::nullopt);
|
||||
|
||||
// Register 0-18 + 29 + 30 are caller saved
|
||||
static constexpr uint32_t CALLER_GPR_MASK = 0b0110'0000'0000'0111'1111'1111'1111'1111U;
|
||||
@@ -217,9 +167,7 @@ protected:
|
||||
if (SupportsPreserveAllABI) {
|
||||
return SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
} else {
|
||||
SpillStaticRegs(TmpReg, {
|
||||
.FPRs = FPRs,
|
||||
});
|
||||
SpillStaticRegs(TmpReg, FPRs);
|
||||
return PushDynamicRegs(TmpReg);
|
||||
}
|
||||
}
|
||||
@@ -229,7 +177,7 @@ protected:
|
||||
FillForPreserveAllABICall(FPRs);
|
||||
} else {
|
||||
PopDynamicRegs();
|
||||
FillStaticRegs({.FPRs = FPRs});
|
||||
FillStaticRegs(FPRs);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,18 +1,24 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "FEXCore/Config/Config.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/PrctlUtils.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
#include "LookupCache.h"
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <sys/prctl.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
|
||||
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
|
||||
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
|
||||
@@ -34,8 +40,6 @@ namespace CPU {
|
||||
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
|
||||
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB
|
||||
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB_UPPER
|
||||
{0x0706'0504'0302'0100ULL, 0x1716'1514'1312'1110ULL}, // NAMED_VECTOR_256_MID_ELEMENT_SWAP
|
||||
{0x0F0E'0D0C'0B0A'0908ULL, 0x1F1E'1D1C'1B1A'1918ULL}, // NAMED_VECTOR_256_MID_ELEMENT_SWAP_UPPER
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_ONE
|
||||
{0xD49A'784B'CD1B'8AFEULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_LOG2_10
|
||||
{0xB8AA'3B29'5C17'F0BCULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_LOG2_E
|
||||
@@ -266,52 +270,52 @@ namespace CPU {
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
CPUBackend::CPUBackend(SharedCodeBufferManager& SharedCodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
|
||||
CPUBackend::CPUBackend(CodeBufferManager& CodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
|
||||
: ThreadState(ThreadState)
|
||||
, SharedCodeBuffers(SharedCodeBuffers) {
|
||||
, CodeBuffers(CodeBuffers) {
|
||||
|
||||
auto& Ptrs = ThreadState->CurrentFrame->Pointers;
|
||||
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
// Initialize named vector constants.
|
||||
for (size_t i = 0; i < FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX; ++i) {
|
||||
Ptrs.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
|
||||
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
|
||||
}
|
||||
|
||||
// Copy named vector constants.
|
||||
memcpy(Ptrs.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
|
||||
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
|
||||
|
||||
// Initialize Indexed named vector constants.
|
||||
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] =
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] =
|
||||
reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
|
||||
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] =
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] =
|
||||
reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
|
||||
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] =
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] =
|
||||
reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
|
||||
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] =
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] =
|
||||
reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
|
||||
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] =
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] =
|
||||
reinterpret_cast<uint64_t>(DPPS_MASK.data());
|
||||
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] =
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] =
|
||||
reinterpret_cast<uint64_t>(DPPD_MASK.data());
|
||||
Ptrs.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] =
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] =
|
||||
reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
|
||||
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
// Fill in telemetry values
|
||||
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
|
||||
auto& Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
|
||||
Ptrs.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(&Telem);
|
||||
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(&Telem);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
CPUBackend::~CPUBackend() = default;
|
||||
|
||||
auto CPUBackend::AcquireNewSharedCodeBuffer() -> CodeBuffer* {
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer = SharedCodeBuffers.StartLargerCodeBuffer();
|
||||
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
|
||||
|
||||
RegisterForSignalHandler(std::move(PrevCodeBuffer));
|
||||
return CurrentCodeBuffer.get();
|
||||
@@ -329,7 +333,7 @@ namespace CPU {
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
|
||||
auto NewCodeBuffer = SharedCodeBuffers.GetLatest();
|
||||
auto NewCodeBuffer = CodeBuffers.GetLatest();
|
||||
if (CurrentCodeBuffer != NewCodeBuffer) {
|
||||
RegisterForSignalHandler(CurrentCodeBuffer);
|
||||
return std::exchange(CurrentCodeBuffer, NewCodeBuffer);
|
||||
@@ -337,17 +341,96 @@ namespace CPU {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
GuestToHostMap& GetLookupCache(const CodeBuffer& Buffer) {
|
||||
return *Buffer.LookupCache;
|
||||
}
|
||||
|
||||
CodeBuffer::CodeBuffer(size_t Size)
|
||||
: Size(Size) {
|
||||
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
|
||||
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::None)) {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
|
||||
CodeBuffer::~CodeBuffer() {
|
||||
FEXCore::Allocator::VirtualFree(Ptr, Size);
|
||||
}
|
||||
|
||||
auto CodeBufferManager::AllocateNew(size_t Size) -> fextl::shared_ptr<CodeBuffer> {
|
||||
#ifndef _WIN32
|
||||
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
|
||||
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
|
||||
//
|
||||
// MDWE prevents applications from creating RWX memory mappings.
|
||||
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
|
||||
//
|
||||
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
|
||||
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
|
||||
//
|
||||
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
|
||||
//
|
||||
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
|
||||
// -1: The kernel doesn't support MDWE
|
||||
// 0: MDWE is supported but disabled
|
||||
// >0: MDWE is enabled, hence prohibiting RWX mappings
|
||||
#ifndef PR_GET_MDWE
|
||||
#define PR_GET_MDWE 66
|
||||
#endif
|
||||
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
|
||||
if (MDWE != -1 && MDWE != 0) {
|
||||
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
|
||||
}
|
||||
#endif
|
||||
|
||||
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
|
||||
|
||||
Latest = Buffer;
|
||||
LatestOffset = 0;
|
||||
|
||||
OnCodeBufferAllocated(*Buffer);
|
||||
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::GetLatest() {
|
||||
if (!Latest) {
|
||||
AllocateNew(INITIAL_CODE_SIZE);
|
||||
}
|
||||
return Latest;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::StartLargerCodeBuffer() {
|
||||
if (!Latest) {
|
||||
// Allocate initial CodeBuffer and return it
|
||||
return GetLatest();
|
||||
}
|
||||
|
||||
auto NewCodeBufferSize = GetLatest()->Size;
|
||||
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
|
||||
return AllocateNew(NewCodeBufferSize);
|
||||
}
|
||||
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
const auto CheckCodeBuffer = [](const CodeBuffer& Buffer, uintptr_t Address) {
|
||||
const auto BufferPtr = reinterpret_cast<uintptr_t>(Buffer.GetBufferBase());
|
||||
const uintptr_t LastPageAddr = BufferPtr + Buffer.UsableSize();
|
||||
return (Address >= BufferPtr && Address < LastPageAddr);
|
||||
auto CheckCodeBuffer = [](CodeBuffer& Buffer, uintptr_t Address) {
|
||||
// The last page of the code buffer is protected, so we need to exclude it from the valid range
|
||||
// when checking if the address is in the code buffer.
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
return (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr);
|
||||
};
|
||||
|
||||
if (CheckCodeBuffer(*CurrentCodeBuffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
for (const auto& Buffer : SignalHandlerCodeBuffers) {
|
||||
for (auto& Buffer : SignalHandlerCodeBuffers) {
|
||||
if (CheckCodeBuffer(*Buffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -8,8 +8,6 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/SharedCodeBufferManager.h"
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -18,7 +16,6 @@ $end_info$
|
||||
#include <FEXCore/fextl/map.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
@@ -44,10 +41,58 @@ namespace CodeSerialize {
|
||||
struct GuestToHostMap;
|
||||
|
||||
namespace CPU {
|
||||
struct CodeBuffer {
|
||||
uint8_t* Ptr;
|
||||
size_t Size;
|
||||
|
||||
fextl::unique_ptr<GuestToHostMap> LookupCache;
|
||||
|
||||
CodeBuffer(size_t Size);
|
||||
CodeBuffer(const CodeBuffer&) = delete;
|
||||
CodeBuffer& operator=(const CodeBuffer&) = delete;
|
||||
CodeBuffer(CodeBuffer&& oth) = delete;
|
||||
CodeBuffer& operator=(CodeBuffer&&) = delete;
|
||||
|
||||
~CodeBuffer();
|
||||
};
|
||||
|
||||
/**
|
||||
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
|
||||
*
|
||||
* The CodeBuffer is managed as a partially persistent data structure:
|
||||
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
|
||||
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
|
||||
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
|
||||
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
|
||||
*/
|
||||
class CodeBufferManager {
|
||||
public:
|
||||
// Get the CodeBuffer that was most recently allocated.
|
||||
// This is the only CodeBuffer that data may be written to.
|
||||
fextl::shared_ptr<CodeBuffer> GetLatest();
|
||||
|
||||
// Allocate a new CodeBuffer with geometric growth up to an internal maximum.
|
||||
// Subsequent calls to GetLatest will point to the returned buffer.
|
||||
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer();
|
||||
|
||||
// Write offset into the latest CodeBuffer
|
||||
std::size_t LatestOffset {};
|
||||
|
||||
// Protects writes to the latest CodeBuffer and changes to LatestOffset
|
||||
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
|
||||
|
||||
virtual void OnCodeBufferAllocated(CodeBuffer&) {};
|
||||
|
||||
private:
|
||||
fextl::shared_ptr<CodeBuffer> Latest;
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
|
||||
};
|
||||
|
||||
class CPUBackend {
|
||||
public:
|
||||
|
||||
CPUBackend(SharedCodeBufferManager&, FEXCore::Core::InternalThreadState*);
|
||||
CPUBackend(CodeBufferManager&, FEXCore::Core::InternalThreadState*);
|
||||
|
||||
virtual ~CPUBackend();
|
||||
|
||||
@@ -57,8 +102,6 @@ namespace CPU {
|
||||
fextl::map<uint64_t, uint8_t*> EntryPoints;
|
||||
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
|
||||
size_t Size;
|
||||
// Offset of BlockBegin from the start of the CodeBuffer it lives in
|
||||
uint64_t HostCodeOffset;
|
||||
};
|
||||
|
||||
// Header that can live at the start of a JIT block.
|
||||
@@ -118,11 +161,7 @@ namespace CPU {
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
virtual CompiledCode LoadCachedCode(std::span<const uint8_t> HostBytes) {
|
||||
return {};
|
||||
}
|
||||
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) = 0;
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
@@ -145,9 +184,8 @@ namespace CPU {
|
||||
|
||||
FEXCore::Core::InternalThreadState* ThreadState;
|
||||
|
||||
// Acquires a new shared code buffer, setting `CurrentCodeBuffer` and returning a pointer to it.
|
||||
[[nodiscard]]
|
||||
CodeBuffer* AcquireNewSharedCodeBuffer();
|
||||
CodeBuffer* GetEmptyCodeBuffer();
|
||||
|
||||
// This is the code buffer containing the main code under execution by this thread.
|
||||
// CheckCodeBufferUpdate must be used before compiling new code.
|
||||
@@ -156,7 +194,7 @@ namespace CPU {
|
||||
// Old CodeBuffer generations required to be valid until returning from signal handlers
|
||||
fextl::vector<fextl::shared_ptr<CodeBuffer>> SignalHandlerCodeBuffers;
|
||||
|
||||
SharedCodeBufferManager& SharedCodeBuffers;
|
||||
CodeBufferManager& CodeBuffers;
|
||||
|
||||
private:
|
||||
void RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer>);
|
||||
|
||||
@@ -14,7 +14,6 @@ $end_info$
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
@@ -24,7 +23,7 @@ $end_info$
|
||||
|
||||
namespace FEXCore {
|
||||
namespace ProductNames {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
#ifdef _M_ARM_64
|
||||
static const char ARM_UNKNOWN[] = "Unknown ARM CPU";
|
||||
static const char ARM_A57[] = "Cortex-A57";
|
||||
static const char ARM_A72[] = "Cortex-A72";
|
||||
@@ -89,18 +88,16 @@ namespace ProductNames {
|
||||
static const char ARM_Blizzard_M2Pro[] = "Apple Blizzard (M2 Pro)";
|
||||
static const char ARM_Avalanche_M2Max[] = "Apple Avalanche (M2 Max)";
|
||||
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
|
||||
static const char ARM_AppleSilicon[] = "Apple Silicon";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
static const char ARM_ORYON_3[] = "Oryon-3";
|
||||
static const char ARM_Ampere_1[] = "AmpereOne";
|
||||
static const char ARM_Ampere_1A[] = "AmpereOneA";
|
||||
static const char ARM_Ampere_1B[] = "AmpereOneB";
|
||||
static const char ARM_Ampere_1C[] = "AmpereOneC";
|
||||
#else
|
||||
#endif
|
||||
} // namespace ProductNames
|
||||
|
||||
static uint32_t GetCPUID_Syscall() {
|
||||
uint32_t GetCPUID_Syscall() {
|
||||
uint32_t CPU {};
|
||||
FHU::Syscalls::getcpu(&CPU, nullptr);
|
||||
return CPU;
|
||||
@@ -141,21 +138,20 @@ constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
|
||||
});
|
||||
#endif
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
uint64_t GetCycleCounterFrequency() {
|
||||
#ifdef _M_ARM_64
|
||||
uint32_t GetCycleCounterFrequency() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], CNTFRQ_EL0" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static uint32_t GetCPUID_TPIDRRO() {
|
||||
uint32_t GetCPUID_TPIDRRO() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], TPIDRRO_EL0" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
FEX_CONFIG_OPT(HideHybrid, HIDEHYBRID);
|
||||
PerCPUData.resize(Cores);
|
||||
|
||||
uint64_t MIDR {};
|
||||
@@ -172,11 +168,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
|
||||
if (HideHybrid()) {
|
||||
// Hide the hybrid flag.
|
||||
Hybrid = false;
|
||||
}
|
||||
|
||||
struct CPUMIDR {
|
||||
uint8_t Implementer;
|
||||
uint16_t Part;
|
||||
@@ -187,9 +178,8 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 68> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 66> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x51, 0x002, 1, ProductNames::ARM_ORYON_3}, // Qualcomm Oryon-3
|
||||
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
|
||||
|
||||
{0x61, 0x039, 1, ProductNames::ARM_Avalanche_M2Max}, // Apple Avalanche (M2 Max)
|
||||
@@ -198,7 +188,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x61, 0x029, 1, ProductNames::ARM_Firestorm_M1Max}, // Apple Firestorm (M1 Max)
|
||||
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
|
||||
{0x61, 0, 1, ProductNames::ARM_AppleSilicon}, // QEmu Apple Silicon
|
||||
|
||||
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
|
||||
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
|
||||
@@ -237,7 +226,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0xc0, 0xac3, 1, ProductNames::ARM_Ampere_1}, // AmpereOne
|
||||
{0xc0, 0xac4, 1, ProductNames::ARM_Ampere_1A}, // AmpereOneA
|
||||
{0xc0, 0xac5, 1, ProductNames::ARM_Ampere_1B}, // AmpereOneB
|
||||
{0xc0, 0xac7, 1, ProductNames::ARM_Ampere_1C}, // AmpereOneC
|
||||
|
||||
{0x4e, 0x010, 1, ProductNames::ARM_Olympus}, // Olympus
|
||||
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
|
||||
@@ -316,8 +304,9 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
// Walk our list of CPUMIDRs to find the most little core
|
||||
for (size_t j = LowestMIDRIdx; j < CPUMIDRs.size(); ++j) {
|
||||
const auto& MIDROption = CPUMIDRs[j];
|
||||
auto& MIDROption = CPUMIDRs[i];
|
||||
if ((MIDROption.Implementer == Implementer && MIDROption.Part == Part) || (MIDROption.Implementer == 0 && MIDROption.Part == 0)) {
|
||||
|
||||
LowestMIDRIdx = j;
|
||||
LowestMIDR = MIDR;
|
||||
break;
|
||||
@@ -394,8 +383,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
} else {
|
||||
// If we aren't hybrid then just claim everything is big
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
const auto MIDRIndex = HideHybrid() ? 0 : i;
|
||||
uint32_t MIDR = PerCPUData[MIDRIndex].MIDR;
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
|
||||
PerCPUData[i].IsBig = true;
|
||||
@@ -409,7 +397,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
}
|
||||
|
||||
#else
|
||||
uint64_t GetCycleCounterFrequency() {
|
||||
uint32_t GetCycleCounterFrequency() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -453,10 +441,10 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(GetCPUID() << 24); // Local APIC ID
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(0 << 24); // Local APIC ID
|
||||
|
||||
Res.ecx = (1 << 0) | // SSE3
|
||||
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
|
||||
@@ -493,8 +481,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual 8086 mode enhancements
|
||||
(1 << 2) | // Debugging extensions
|
||||
(1 << 3) | // Page size extension
|
||||
(0 << 2) | // Debugging extensions
|
||||
(0 << 3) | // Page size extension
|
||||
(1 << 4) | // RDTSC supported
|
||||
(1 << 5) | // MSR supported
|
||||
(1 << 6) | // PAE
|
||||
@@ -519,7 +507,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
(1 << 25) | // SSE
|
||||
(1 << 26) | // SSE2
|
||||
(0 << 27) | // Self Snoop
|
||||
(0 << 28) | // (HTT) Max APIC IDs reserved field is valid
|
||||
(1 << 28) | // Max APIC IDs reserved field is valid
|
||||
(1 << 29) | // Thermal monitor
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Pending break enable
|
||||
@@ -649,13 +637,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
if (Leaf == 0) {
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t SUPPORTS_RDPID = 1;
|
||||
#else
|
||||
// RDPID under WIN32 is only supported if CPUIndex is available in TPIDRRO.
|
||||
const uint32_t SUPPORTS_RDPID = SupportsCPUIndexInTPIDRRO;
|
||||
#endif
|
||||
|
||||
// Disable Enhanced REP MOVS when TSO is enabled.
|
||||
// vcruntime140 memmove will use `rep movsb` in this case which completely destroys perf in Hades(appId 1145360)
|
||||
// This is due to LRCPC performance on Cortex being abysmal.
|
||||
@@ -721,7 +702,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 19) | // MPX MAWAU
|
||||
(0 << 20) | // MPX MAWAU
|
||||
(0 << 21) | // MPX MAWAU
|
||||
(SUPPORTS_RDPID << 22) | // RDPID Read Processor ID
|
||||
(1 << 22) | // RDPID Read Processor ID
|
||||
(0 << 23) | // AES Key Locker
|
||||
(1 << 24) | // bus-lock-detect
|
||||
(0 << 25) | // CLDEMOTE
|
||||
@@ -764,95 +745,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 29) | // Arch capabilities - Speculative side channel mitigations
|
||||
(0 << 30) | // Arch capabilities - MSR module specific
|
||||
(0 << 31); // SSBD - Speculative Store Bypass Disable
|
||||
} else if (Leaf == 1) {
|
||||
Res.eax = (0U << 0) | // SHA512
|
||||
(0U << 1) | // SM3
|
||||
(0U << 2) | // SM4
|
||||
(0U << 3) | // RAO_INT
|
||||
(0U << 4) | // AVX_VNNI
|
||||
(0U << 5) | // AVX512_BF16
|
||||
(0U << 6) | // LASS (Linear Address Space Separation)
|
||||
(0U << 7) | // CMPCCXADD
|
||||
(0U << 8) | // ARCH_PERFMON_EXT
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // FAST_REP_MOVSB
|
||||
(0U << 11) | // FAST_REP_STOSB
|
||||
(0U << 12) | // FAST_REP_CMPSB_SCASB
|
||||
(0U << 13) | // Reserved
|
||||
(0U << 14) | // Reserved
|
||||
(0U << 15) | // Reserved
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // FRED (Flexible Return and Event Delivery)
|
||||
(0U << 18) | // LKGS (Load into Kernel GS Base)
|
||||
(0U << 19) | // WRMSRNS
|
||||
(0U << 20) | // NMI_SRC
|
||||
(0U << 21) | // AMX_FP16
|
||||
(0U << 22) | // HRESET
|
||||
(0U << 23) | // AVX_IFMA
|
||||
(0U << 24) | // Reserved
|
||||
(0U << 25) | // Reserved
|
||||
(0U << 26) | // LAM (Linear Address Masking)
|
||||
(0U << 27) | // MSRLIST
|
||||
(0U << 28) | // Reserved
|
||||
(0U << 29) | // Reserved
|
||||
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
|
||||
(0U << 31); // MOVRS
|
||||
|
||||
// Bits 4-31 currently reserved.
|
||||
Res.ebx = (0U << 0) | // PPIN
|
||||
(0U << 1) | // PBNDKB
|
||||
(0U << 2) | // Reserved
|
||||
(0U << 3); // CPUIDMAXVAL_LIM_RMV
|
||||
|
||||
// Bits 6-31 also reserved.
|
||||
Res.ecx = (0U << 0) | // RDT_M_ASYM
|
||||
(0U << 1) | // RDT_A_ASYM
|
||||
(0U << 2) | // Reserved
|
||||
(0U << 3) | // Reserved
|
||||
(0U << 4) | // Reserved
|
||||
(0U << 5); // MSR_IMM
|
||||
|
||||
// Bits 25-31 also reserved.
|
||||
Res.edx = (0U << 0) | // Reserved
|
||||
(0U << 1) | // Reserved
|
||||
(0U << 2) | // Reserved
|
||||
(0U << 3) | // Reserved
|
||||
(0U << 4) | // AVX_VNNI_INT8
|
||||
(0U << 5) | // AVX_NE_CONVERT
|
||||
(0U << 6) | // Reserved
|
||||
(0U << 7) | // Reserved
|
||||
(0U << 8) | // AMX_COMPLEX
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // AVX_VNNI_INT16
|
||||
(0U << 11) | // Reserved
|
||||
(0U << 12) | // Reserved
|
||||
(0U << 13) | // UTMR (User-timer events)
|
||||
(0U << 14) | // PREFETCHI
|
||||
(0U << 15) | // USER_MSR
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // UIRET_UIF
|
||||
(0U << 18) | // CET_SSS
|
||||
(0U << 19) | // AVX10
|
||||
(0U << 20) | // Reserved
|
||||
(0U << 21) | // APX_F
|
||||
(0U << 22) | // SEC-TEE_ATTESTATION
|
||||
(0U << 23) | // MWAIT
|
||||
(0U << 24); // SLSM (Static LSM)
|
||||
} else if (Leaf == 2) {
|
||||
// All bits are reserved except for EDX
|
||||
Res.eax = 0;
|
||||
Res.ebx = 0;
|
||||
Res.ecx = 0;
|
||||
|
||||
// Bits 8-31 are reserved.
|
||||
Res.edx = (0U << 0) | // PSFD
|
||||
(0U << 1) | // IPRED_CTRL
|
||||
(0U << 2) | // RRSBA_CTRL
|
||||
(0U << 3) | // DDPD_U
|
||||
(0U << 4) | // BHI_CTRL
|
||||
(0U << 5) | // MCDT_NO
|
||||
(0U << 6) | // UC_LOCK_DISABLE
|
||||
(0U << 7); // MONITOR_MITG_NO
|
||||
}
|
||||
|
||||
return Res;
|
||||
@@ -910,7 +802,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
// TSC frequency = ECX * EBX / EAX
|
||||
uint64_t FrequencyHz = GetCycleCounterFrequency();
|
||||
uint32_t FrequencyHz = GetCycleCounterFrequency();
|
||||
if (FrequencyHz) {
|
||||
Res.eax = 1;
|
||||
Res.ebx = 1U << CTX->Config.TSCScale;
|
||||
@@ -931,27 +823,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_1Ah(uint32_t Leaf) const {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_24h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
|
||||
if (Leaf == 0) {
|
||||
// EAX indicates the maximum number of subleaves.
|
||||
Res.eax = 0;
|
||||
|
||||
// Bits 19-31 reserved
|
||||
// NOTE: We return all zero here until we have a CPU with AVX10
|
||||
// even if some of the fields otherwise have fixed values.
|
||||
Res.ebx = (0U << 0) | // (bits 0-7 specify the vector ISA version)
|
||||
(0U << 16); // Defined as always 0b111
|
||||
|
||||
// All bits reserved
|
||||
Res.ecx = 0;
|
||||
Res.edx = 0;
|
||||
}
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
// Hypervisor CPUID information leaf
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0000h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
@@ -983,10 +854,10 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0001h(uint32_t Leaf) con
|
||||
constexpr uint32_t MaximumSubLeafNumber = 2;
|
||||
if (Leaf == 0) {
|
||||
// EAX[3:0] Is the host architecture that FEX is running under
|
||||
#ifdef ARCHITECTURE_x86_64
|
||||
#ifdef _M_X86_64
|
||||
// EAX[3:0] = 1 = x86_64 host architecture
|
||||
Res.eax |= 0b0001;
|
||||
#elif defined(ARCHITECTURE_arm64)
|
||||
#elif defined(_M_ARM_64)
|
||||
// EAX[3:0] = 2 = AArch64 host architecture
|
||||
Res.eax |= 0b0010;
|
||||
#else
|
||||
@@ -1038,38 +909,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
Res.ecx = (1 << 0) | // LAHF/SAHF
|
||||
(1 << 1) | // 0 = Single core product, 1 = multi core product
|
||||
(0 << 2) | // SVM
|
||||
(1 << 3) | // Extended APIC register space
|
||||
(0 << 4) | // LOCK MOV CR0 means MOV CR8
|
||||
(1 << 5) | // ABM instructions
|
||||
(CTX->HostFeatures.SupportsSSE4a << 6) | // SSE4a
|
||||
(0 << 7) | // Misaligned SSE mode
|
||||
(1 << 8) | // PREFETCHW
|
||||
(0 << 9) | // OS visible workaround support
|
||||
(0 << 10) | // Instruction based sampling support
|
||||
(0 << 11) | // XOP
|
||||
(0 << 12) | // SKINIT
|
||||
(0 << 13) | // Watchdog timer support
|
||||
(0 << 14) | // Reserved
|
||||
(0 << 15) | // Lightweight profiling support
|
||||
(0 << 16) | // FMA4
|
||||
(1 << 17) | // Translation cache extension
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // XOP-TBM
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // MONITORX
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
Res.ecx = (1 << 0) | // LAHF/SAHF
|
||||
(1 << 1) | // 0 = Single core product, 1 = multi core product
|
||||
(0 << 2) | // SVM
|
||||
(1 << 3) | // Extended APIC register space
|
||||
(0 << 4) | // LOCK MOV CR0 means MOV CR8
|
||||
(1 << 5) | // ABM instructions
|
||||
(0 << 6) | // SSE4a
|
||||
(0 << 7) | // Misaligned SSE mode
|
||||
(1 << 8) | // PREFETCHW
|
||||
(0 << 9) | // OS visible workaround support
|
||||
(0 << 10) | // Instruction based sampling support
|
||||
(0 << 11) | // XOP
|
||||
(0 << 12) | // SKINIT
|
||||
(0 << 13) | // Watchdog timer support
|
||||
(0 << 14) | // Reserved
|
||||
(0 << 15) | // Lightweight profiling support
|
||||
(0 << 16) | // FMA4
|
||||
(1 << 17) | // Translation cache extension
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // XOP-TBM
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // MONITORX
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual mode extensions
|
||||
@@ -1097,7 +968,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
(1 << 23) | // MMX
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(1 << 26) | // 1 gigabit pages
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(SUPPORTS_RDTSCP << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
@@ -1223,9 +1094,9 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) con
|
||||
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
|
||||
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
(std::bit_ceil(Cores) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
(CoreCount << 0); // Count count subtract one
|
||||
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
(CoreCount << 0); // Count count subtract one
|
||||
|
||||
return Res;
|
||||
}
|
||||
@@ -1347,7 +1218,7 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
|
||||
CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx}
|
||||
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO != 0}
|
||||
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO}
|
||||
, GetCPUID {GetCPUID_Syscall} {
|
||||
Cores = CTX->HostFeatures.CPUMIDRs.size();
|
||||
|
||||
@@ -1356,7 +1227,7 @@ CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
|
||||
|
||||
SetupFeatures();
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
#ifdef _M_ARM_64
|
||||
if (SupportsCPUIndexInTPIDRRO) {
|
||||
GetCPUID = GetCPUID_TPIDRRO;
|
||||
}
|
||||
|
||||
@@ -14,7 +14,7 @@ namespace Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
uint64_t GetCycleCounterFrequency();
|
||||
uint32_t GetCycleCounterFrequency();
|
||||
|
||||
// Debugging define to switch what family of CPU we execute as.
|
||||
// Might be useful if an application makes an assumption about a CPU.
|
||||
@@ -159,7 +159,7 @@ private:
|
||||
|
||||
struct CPUData {
|
||||
const char* ProductName {};
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
#ifdef _M_ARM_64
|
||||
uint32_t MIDR {};
|
||||
#endif
|
||||
bool IsBig {};
|
||||
@@ -176,7 +176,6 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_24h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0001h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf) const;
|
||||
@@ -201,7 +200,7 @@ private:
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
void SetupFeatures();
|
||||
static constexpr size_t PRIMARY_FUNCTION_COUNT = 37;
|
||||
static constexpr size_t PRIMARY_FUNCTION_COUNT = 27;
|
||||
static constexpr size_t HYPERVISOR_FUNCTION_COUNT = 2;
|
||||
static constexpr size_t EXTENDED_FUNCTION_COUNT = 32;
|
||||
static constexpr std::array<FunctionHandler, PRIMARY_FUNCTION_COUNT> Primary = {
|
||||
@@ -269,48 +268,7 @@ private:
|
||||
#ifndef CPUID_AMD
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
&CPUIDEmu::Function_1Ah,
|
||||
// 0x1B: PCONFIG info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1C: Last Branch Records (LBR) info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1D: Tile info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1E: TMUL info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1F: V2 Extended topology
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x20: Processor History Reset info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x21: Unimplemented
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x22: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x23: Architectural Performance Monitoring Extended
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x24: Converged Vector ISA
|
||||
&CPUIDEmu::Function_24h,
|
||||
#else
|
||||
// 0x1A: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1B: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1C: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1D: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1E: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1F: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x20: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x21: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x22: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x23: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x24: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#endif
|
||||
};
|
||||
@@ -319,7 +277,7 @@ private:
|
||||
// 0: Highest function parameter and ID
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 1: Processor info
|
||||
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 2: Cache and TLB info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 3: Serial Number(previously), now reserved
|
||||
@@ -382,49 +340,9 @@ private:
|
||||
#ifndef CPUID_AMD
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1B: PCONFIG info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1C: Last Branch Records (LBR) info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1D: Tile info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1E: TMUL info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1F: V2 Extended topology
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x20: Processor History Reset info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x21: Unimplemented/Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x22: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x23: Architectural Performance Monitoring Extended
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x24: Converged Vector ISA
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
|
||||
#else
|
||||
// 0x1A: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1B: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1C: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1D: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1E: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1F: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x20: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x21: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x22: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x23: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x24: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#endif
|
||||
}};
|
||||
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -9,9 +9,6 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#ifdef ZYDIS_DISASSEMBLER
|
||||
#include <Zydis/Zydis.h>
|
||||
#endif
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
@@ -30,7 +27,7 @@ $end_info$
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include "Utils/variable_length_integer.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -76,9 +73,6 @@ $end_info$
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
#include <arm_acle.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
@@ -106,8 +100,6 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
DiskCache.Init(this);
|
||||
}
|
||||
|
||||
struct GetFrameBlockInfoResult {
|
||||
@@ -347,18 +339,13 @@ void ContextImpl::SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState
|
||||
}
|
||||
|
||||
bool ContextImpl::InitCore() {
|
||||
if (CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHINGWIP()) {
|
||||
// Start with a larger code buffer to avoid resizes that would discard code
|
||||
StartMaximalCodeBuffer();
|
||||
}
|
||||
|
||||
// Initialize the CPU core signal handlers & DispatcherConfig
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
|
||||
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
SignalDelegation->SetConfig(Dispatcher->MakeSignalDelegatorConfig());
|
||||
|
||||
#if defined(_WIN32) && !defined(ARCHITECTURE_arm64ec)
|
||||
#if defined(_WIN32) && !defined(_M_ARM_64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
@@ -368,16 +355,6 @@ bool ContextImpl::InitCore() {
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
}
|
||||
|
||||
if constexpr (BLOCK_DEBUGGING) {
|
||||
// If the developer wants to do any single-stepping points or watch points.
|
||||
// Add them here.
|
||||
//
|
||||
// eg:
|
||||
// BlockDebuggerTracker.AllTargetSingleStep();
|
||||
// BlockDebuggerTracker.AddSingleStepTarget(0x14000'0000ULL);
|
||||
// BlockDebuggerTracker.AddWriteWatchPoint(0x420BA5ED);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -386,9 +363,6 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
|
||||
}
|
||||
|
||||
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Update the thread pointer for Thunk return to the latest.
|
||||
Thread->CurrentFrame->Pointers.ThunkCallbackRet = SignalDelegation->GetThunkCallbackRET();
|
||||
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
// If it is the parent thread that died then just leave
|
||||
@@ -396,32 +370,37 @@ void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this, Thread);
|
||||
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
|
||||
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
|
||||
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>(this);
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
|
||||
|
||||
Thread->CurrentFrame->State.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->State.L1Mask = Thread->LookupCache->GetScaledL1PointerMask();
|
||||
|
||||
Thread->CurrentFrame->Pointers.L2Pointer = Thread->LookupCache->GetPagePointer();
|
||||
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
|
||||
|
||||
Dispatcher->InitThreadPointers(Thread);
|
||||
|
||||
Thread->PassManager->AddDefaultPasses(this);
|
||||
Thread->PassManager->AddDefaultValidationPasses();
|
||||
|
||||
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
|
||||
// Create CPU backend
|
||||
Thread->PassManager->InsertRegisterAllocationPass(this);
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
|
||||
// We finalize *after* the CPU backend is initialized, as the CPU backend will
|
||||
// provide necessary register information to the register allocation pass.
|
||||
Thread->PassManager->Finalize();
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(const FEXCore::Core::CPUState* NewThreadState) {
|
||||
FEXCore::Core::InternalThreadState*
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState) {
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
|
||||
.CTX = this,
|
||||
};
|
||||
FEXCore::Allocator::VirtualName("FEXMem_ThreadState", Thread, sizeof(*Thread));
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
if (NewThreadState) {
|
||||
@@ -455,10 +434,6 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
|
||||
Profiler::PostForkAction(Child);
|
||||
if (Child) {
|
||||
if (CodeMapWriter) {
|
||||
CodeMapWriter->ResetAfterFork();
|
||||
}
|
||||
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
StrictSplitLockMutex = 0;
|
||||
@@ -468,6 +443,7 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
FEXCore::Utils::SpinWaitLock::unlock(&StrictSplitLockMutex);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -480,14 +456,9 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::OnCodeBufferAllocated(const fextl::shared_ptr<CPU::CodeBuffer>& Buffer) {
|
||||
void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
|
||||
if (Config.GlobalJITNaming()) {
|
||||
Symbols.RegisterJITSpace(Buffer->GetBufferBase(), Buffer->TotalAllocationSize());
|
||||
}
|
||||
|
||||
{
|
||||
std::scoped_lock lk {CodeBufferListLock};
|
||||
CodeBufferList.emplace_back(Buffer);
|
||||
Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -499,28 +470,24 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, boo
|
||||
Thread->CPUBackend->ClearCache();
|
||||
} else {
|
||||
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
Thread->LookupCache->ClearCache(lk);
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
|
||||
FEXCore::File::File FD = FEXCore::File::File::GetStdERR();
|
||||
fextl::ostringstream out;
|
||||
fextl::stringstream out;
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR);
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", NewIR.PostRA() ? "post" : "pre", GuestRIP, out.str());
|
||||
}
|
||||
|
||||
bool ContextImpl::CheckIfBlockIsCacheable(FEXCore::Core::InternalThreadState& Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
return Thread.FrontendDecoder->CheckIfCacheable(Thread, reinterpret_cast<const uint8_t*>(GuestRIP), GuestRIP, MaxInst);
|
||||
}
|
||||
};
|
||||
|
||||
ContextImpl::GenerateIRResult
|
||||
ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
uint64_t TotalInstructions {0};
|
||||
@@ -540,46 +507,29 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
if (!HasCustomIR) {
|
||||
const auto* GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
|
||||
const uint8_t* GuestCode {};
|
||||
GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
|
||||
|
||||
Thread->FrontendDecoder->DecodeLoop(GuestCode);
|
||||
bool HadDispatchError {false};
|
||||
bool HadInvalidInst {false};
|
||||
|
||||
const auto* BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
const auto& CodeBlocks = BlockInfo->Blocks;
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(Thread, GuestCode, GuestRIP, MaxInst);
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, &CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
|
||||
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
auto CodeBlocks = &BlockInfo->Blocks;
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
|
||||
AreMonoHacksActive() && MonoBackpatcherBlock.load(std::memory_order_relaxed) == GuestRIP);
|
||||
|
||||
const auto GPRSize = Thread->OpDispatcher->GetGPROpSize();
|
||||
|
||||
#ifdef ZYDIS_DISASSEMBLER
|
||||
const auto ZydisMachineMode = Config.Is64BitMode ? ZYDIS_MACHINE_MODE_LONG_64 : ZYDIS_MACHINE_MODE_LEGACY_32;
|
||||
if (FEXCore::Config::Get_X86DISASSEMBLE()) {
|
||||
const uint64_t DecodedMin = Thread->FrontendDecoder->DecodedMinAddress;
|
||||
const uint64_t DecodedMax = Thread->FrontendDecoder->DecodedMaxAddress;
|
||||
LogMan::Msg::IFmt("Guest x86 Begin (RIP={:#x}, {:#x}-{:#x})", GuestRIP, DecodedMin, DecodedMax);
|
||||
}
|
||||
#endif
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks.size(); ++j) {
|
||||
const auto& Block = CodeBlocks[j];
|
||||
|
||||
// Dispatch failures and invalid instructions terminate only the decoded
|
||||
// block that contains them. Other block targets in the same multiblock
|
||||
// compilation unit are independent entry paths.
|
||||
bool HadDispatchError {false};
|
||||
bool HadInvalidInst {false};
|
||||
|
||||
#ifdef ZYDIS_DISASSEMBLER
|
||||
if (FEXCore::Config::Get_X86DISASSEMBLE() && CodeBlocks.size() > 1) {
|
||||
LogMan::Msg::IFmt(" Block {} Entry={:#x} NumInsts={}", j, Block.Entry, Block.NumInstructions);
|
||||
}
|
||||
#endif
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
|
||||
|
||||
bool BlockInForceTSOValidRange = false;
|
||||
auto InstForceTSOIt = ForceTSOInstructions.end();
|
||||
if (ForceTSOValidRanges.Contains({Block.Entry, Block.Entry + Block.Size})) {
|
||||
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); It != ForceTSOInstructions.end() && *It < Block.Entry + Block.Size) {
|
||||
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); *It < Block.Entry + Block.Size) {
|
||||
InstForceTSOIt = It;
|
||||
BlockInForceTSOValidRange = true;
|
||||
}
|
||||
@@ -588,16 +538,18 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
|
||||
uint64_t BlockInstructionsLength {};
|
||||
|
||||
// Reset any block-specific state
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
|
||||
const uint64_t InstsInBlock = Block.NumInstructions;
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
uint64_t BlockInstructionsLength {};
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
uint64_t InstAddress = Block.Entry + BlockInstructionsLength;
|
||||
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
|
||||
@@ -605,19 +557,6 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
TableInfo = Block.DecodedInstructions[i].TableInfo;
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
|
||||
#ifdef ZYDIS_DISASSEMBLER
|
||||
if (FEXCore::Config::Get_X86DISASSEMBLE()) {
|
||||
const uint8_t* InstBytes = reinterpret_cast<const uint8_t*>(InstAddress);
|
||||
ZydisDisassembledInstruction ZydisInst;
|
||||
if (ZYAN_SUCCESS(ZydisDisassembleIntel(ZydisMachineMode, InstAddress, InstBytes, DecodedInfo->InstSize, &ZydisInst))) {
|
||||
LogMan::Msg::IFmt(" {:#x}: {}", InstAddress, ZydisInst.text);
|
||||
} else {
|
||||
LogMan::Msg::IFmt(" {:#x}: (decode failed, {} bytes)", InstAddress, DecodedInfo->InstSize);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
// Do a partial register cache flush before every instruction. This
|
||||
@@ -641,28 +580,9 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL || Block.ForceFullSMCDetection) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint8_t*>(Block.Entry + BlockInstructionsLength);
|
||||
auto InstAddressReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
|
||||
|
||||
auto crc32 = [](const uint8_t* Ptr, size_t Size) -> uint32_t {
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
uint32_t Result {};
|
||||
#define do_crc(type, suffix) \
|
||||
while (Size >= sizeof(type)) { \
|
||||
Result = __crc32##suffix(Result, *reinterpret_cast<const type*>(Ptr)); \
|
||||
Ptr += sizeof(type); \
|
||||
Size -= sizeof(type); \
|
||||
}
|
||||
do_crc(uint64_t, d);
|
||||
do_crc(uint32_t, w);
|
||||
do_crc(uint16_t, h);
|
||||
do_crc(uint8_t, b);
|
||||
return Result;
|
||||
#else
|
||||
// Unsupported on non-arm.
|
||||
return 0;
|
||||
#endif
|
||||
};
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(
|
||||
Thread->OpDispatcher->Constant(crc32(ExistingCodePtr, DecodedInfo->InstSize)), InstAddressReg, DecodedInfo->InstSize);
|
||||
std::array<uint8_t, 0x10> CodeOriginal;
|
||||
memcpy(CodeOriginal.data(), ExistingCodePtr, DecodedInfo->InstSize);
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CodeOriginal, InstAddressReg, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->CondJump(CodeChanged);
|
||||
|
||||
@@ -671,20 +591,13 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->SetTrueJumpTarget(InvalidateCodeCond, CodeWasChangedBlock);
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
|
||||
// Generate a relocatable entry for invalidation purposes.
|
||||
auto EntryReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, 0);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry(EntryReg);
|
||||
|
||||
// Exit the function at this instruction after invalidation.
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
Thread->OpDispatcher->SetFalseJumpTarget(InvalidateCodeCond, NextOpBlock);
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
}
|
||||
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher.OpDispatch) {
|
||||
@@ -729,13 +642,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
|
||||
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::INVALID_INST ||
|
||||
Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::BAD_RELOCATION) {
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
} else if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::UNIMPLEMENTED_INST) {
|
||||
Thread->OpDispatcher->UnimplementedOp(DecodedInfo);
|
||||
} else {
|
||||
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::NOEXEC_INST) {
|
||||
Thread->OpDispatcher->NoExecOp(DecodedInfo);
|
||||
} else {
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -748,8 +658,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
return {std::nullopt, 0, 0, 0, 0};
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return {{}, 0, 0, 0, 0};
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
@@ -766,12 +676,6 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef ZYDIS_DISASSEMBLER
|
||||
if (FEXCore::Config::Get_X86DISASSEMBLE()) {
|
||||
LogMan::Msg::IFmt("Guest x86 End");
|
||||
}
|
||||
#endif
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
@@ -805,10 +709,9 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
MappedSection->FileInfo.SourcecodeMap =
|
||||
SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, CodeMap::GetBaseFilename(MappedSection->FileInfo, false));
|
||||
MappedSection->FileInfo.SourcecodeMap = SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, MappedSection->FileInfo.FileId);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -816,9 +719,6 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, NeedsAddGuestCodeRanges] =
|
||||
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
if (!IRView) {
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
// OpDispatcher IR already released in this case.
|
||||
return {{}, nullptr, 0, 0, false};
|
||||
}
|
||||
|
||||
@@ -828,11 +728,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
// but this would increase lock contention. Redundant frontend runs aren't
|
||||
// as expensive and are easily reverted.
|
||||
if (MaxInst != 1) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
// Raced to compile, release the OpDispatcher IR.
|
||||
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
|
||||
.DebugData = nullptr,
|
||||
.StartAddr = 0,
|
||||
@@ -851,8 +748,6 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
// Release the IR
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
return {
|
||||
.CompiledCode = std::move(CompiledCode),
|
||||
.DebugData = std::move(DebugData),
|
||||
@@ -863,17 +758,6 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
if constexpr (BLOCK_DEBUGGING) {
|
||||
// Block debugging logic is hand-written and needs to be handled with care.
|
||||
// Force MaxInst to only be one in this case.
|
||||
MaxInst = 1;
|
||||
|
||||
// If the entrypoint is part of the single step targets then single step it.
|
||||
if (BlockDebuggerTracker.IsSingleStepTarget(GuestRIP)) {
|
||||
return CompileSingleStep(Frame, GuestRIP);
|
||||
}
|
||||
}
|
||||
|
||||
auto Thread = Frame->Thread;
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedJITTime);
|
||||
@@ -885,54 +769,10 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
Thread->FrontendDecoder->SetupDecodeInstructionsAtEntry(Thread, GuestRIP, MaxInst);
|
||||
|
||||
std::optional<ExecutableFileSectionInfo> Region = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
|
||||
std::optional<DiskCache::CodeHitData> Hit;
|
||||
std::optional<uint64_t> DiskCacheGuestCodeKey;
|
||||
{
|
||||
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedDiskCacheLookupTime);
|
||||
Hit = DiskCache.Lookup(Thread, Region, GuestRIP, DiskCacheGuestCodeKey);
|
||||
if (Hit && !DiskCache.IsValidating()) {
|
||||
auto LoadedCode = Thread->CPUBackend->LoadCachedCode(Hit->HostCode);
|
||||
if (LoadedCode.BlockBegin) {
|
||||
for (auto& CodePage : Hit->GuestPages) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(Thread, Hit->EntryPointRIPs, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Hit->EntryPointRIPs.size() == Hit->EntryPointHostOffsets.size(), "Mismatched Disk Cache entrypoint pairs!");
|
||||
|
||||
uintptr_t CachedHostCode = 0;
|
||||
for (size_t i = 0; i < Hit->EntryPointRIPs.size(); i++) {
|
||||
void* HostAddr = LoadedCode.BlockBegin + Hit->EntryPointHostOffsets[i];
|
||||
Thread->LookupCache->AddBlockMapping(Thread, Hit->EntryPointRIPs[i], Hit->GuestPages, HostAddr);
|
||||
if (Hit->EntryPointRIPs[i] == GuestRIP) {
|
||||
CachedHostCode = reinterpret_cast<uintptr_t>(HostAddr);
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(CachedHostCode != 0, "Couldn't find GuestRIP in Disk Cache entrypoints!");
|
||||
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedDiskCacheHitCount, 1);
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
return CachedHostCode;
|
||||
}
|
||||
}
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedDiskCacheMissCount, 1);
|
||||
}
|
||||
|
||||
// Accumulate a JIT count now, as even if another thread raced us, it should count as a compile.
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedJITCount, 1);
|
||||
|
||||
auto [CompiledCode, DebugData, StartAddr, Length, NeedsAddGuestCodeRanges] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
|
||||
if (CodePtr == nullptr) {
|
||||
@@ -942,18 +782,11 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return reinterpret_cast<uintptr_t>(CodePtr);
|
||||
}
|
||||
|
||||
if (DiskCacheGuestCodeKey && Hit && DiskCache.IsValidating()) {
|
||||
DiskCache.Validate(*DiskCacheGuestCodeKey, *Hit, CompiledCode, Region);
|
||||
}
|
||||
|
||||
// if this ever fires, we need to serialize the offset into disk cache
|
||||
LOGMAN_THROW_A_FMT(StartAddr == GuestRIP, "StartAddr offset from GuestRIP");
|
||||
|
||||
// The core managed to compile the code.
|
||||
if (Config.BlockJITNaming()) {
|
||||
auto FragmentBasePtr = CompiledCode.BlockBegin;
|
||||
|
||||
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
|
||||
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock : DebugData->Subblocks) {
|
||||
@@ -976,7 +809,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
}
|
||||
|
||||
if (Config.LibraryJITNaming() || Config.GDBSymbols()) {
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(Thread, GuestRIP);
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
if (Config.LibraryJITNaming()) {
|
||||
Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, MappedSection->FileInfo.Filename);
|
||||
@@ -988,51 +821,26 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
}
|
||||
}
|
||||
|
||||
fextl::vector<uint64_t> CodePages;
|
||||
|
||||
if (NeedsAddGuestCodeRanges) {
|
||||
// Track in the guest to host map all entrypoints for all pages the compiled block touches, if any page didn't previously
|
||||
// contain code, inform the frontend so it can setup SMC detection.
|
||||
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
CodePages.reserve(BlockInfo->CodePages.size());
|
||||
CodePages.insert(CodePages.end(), BlockInfo->CodePages.begin(), BlockInfo->CodePages.end());
|
||||
for (auto CodePage : BlockInfo->CodePages) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(Thread, BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Disk Cache
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
if (DiskCacheGuestCodeKey) {
|
||||
std::span<const FEXCore::CPU::Relocation> Relocations;
|
||||
if (DebugData && DebugData->Relocations) {
|
||||
Relocations = *DebugData->Relocations;
|
||||
}
|
||||
std::span<const uint8_t> GuestCode = {reinterpret_cast<const uint8_t*>(StartAddr), Length};
|
||||
const Frontend::Decoder::DecodedBlockInformation* BlockInfo =
|
||||
NeedsAddGuestCodeRanges ? Thread->FrontendDecoder->GetDecodedBlockInfo() : nullptr;
|
||||
DiskCache.Store(Thread, Region, GuestRIP, *DiskCacheGuestCodeKey, GuestCode, CompiledCode, Relocations, BlockInfo);
|
||||
}
|
||||
|
||||
if (CodeMapWriter && Region && Region->FileStartVA != 0) {
|
||||
CodeMapWriter->AppendBlock(*Region, GuestRIP);
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
|
||||
Thread->LookupCache->AddBlockMapping(Thread, GuestAddr, CodePages, HostAddr);
|
||||
}
|
||||
|
||||
// Clear any relocations that might have been generated
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
Thread->FrontendDecoder->ValidateDisownedOrFree();
|
||||
Thread->OpDispatcher->ValidateDisownedOrFree();
|
||||
if (NeedsAddGuestCodeRanges) {
|
||||
// Track in the guest to host map all entrypoints for all pages the compiled block touches, if any page didn't previously
|
||||
// contain code, inform the frontend so it can setup SMC detection.
|
||||
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
for (auto CodePage : BlockInfo->CodePages) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
|
||||
Thread->LookupCache->AddBlockMapping(GuestAddr, HostAddr);
|
||||
}
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
@@ -1046,7 +854,6 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->FrontendDecoder->SetupDecodeInstructionsAtEntry(Thread, GuestRIP, 1);
|
||||
auto [CompiledCode, DebugData, StartAddr, Length, _] = CompileCode(Thread, GuestRIP, 1);
|
||||
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
|
||||
if (CodePtr == nullptr) {
|
||||
@@ -1059,37 +866,49 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) {
|
||||
FEXCORE_PROFILE_SCOPED("InvalidateCodeBuffersCodeRange");
|
||||
|
||||
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
std::scoped_lock lk {CodeBufferListLock};
|
||||
auto it = CodeBufferList.begin();
|
||||
while (it != CodeBufferList.end()) {
|
||||
if (auto Strong = it->lock()) {
|
||||
Strong->LookupCache->InvalidateRange(Start, Length);
|
||||
it++;
|
||||
} else {
|
||||
it = CodeBufferList.erase(it);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
|
||||
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
|
||||
uint64_t Start, uint64_t Length) {
|
||||
// Ensures now-modified mappings aren't cached as being in their previous non-executable state.
|
||||
// Accessing FrontendDecoder is safe as the thread's code invalidation mutex must be locked here.
|
||||
Thread->FrontendDecoder->ResetExecutableRangeCache();
|
||||
|
||||
if (Thread->LookupCache->InvalidateCacheRange(Start, Length)) {
|
||||
FEXCORE_PROFILE_SCOPED("InvalidateCallRet");
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
auto& CodePages = Thread->LookupCache->Shared->CodePages;
|
||||
|
||||
auto lower = CodePages.lower_bound(Start >> 12);
|
||||
auto upper = CodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
Accumulator.emplace_back(std::move(it->second));
|
||||
}
|
||||
|
||||
bool InvalidatedAnyEntries = false;
|
||||
for (const auto& PageEntries : Accumulator) {
|
||||
for (const auto& Entry : PageEntries) {
|
||||
if (ContextImpl::ThreadRemoveCodeEntry(Thread, Entry)) {
|
||||
InvalidatedAnyEntries = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (InvalidatedAnyEntries) {
|
||||
// This may cause access violations in the thread on Windows as zeroing is not atomic, this is handled by the frontend
|
||||
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
|
||||
uint64_t Start, uint64_t Length) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
|
||||
}
|
||||
|
||||
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
|
||||
"be unique_locked here");
|
||||
|
||||
return Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
static_cast<ContextImpl*>(Frame->Thread->CTX)->SyscallHandler->InvalidateGuestCodeRange(Frame->Thread, GuestRIP, 1);
|
||||
}
|
||||
@@ -1131,13 +950,12 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
|
||||
const auto GPRSize = this->Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
|
||||
|
||||
// Thunk entry-points don't get cached, don't need to be padded.
|
||||
if (GPRSize == IR::OpSize::i64Bit) {
|
||||
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
|
||||
R->Reg = IR::PhysicalRegister(IR::RegClass::GPRFixed, X86State::REG_R11).Raw;
|
||||
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
|
||||
} else {
|
||||
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
|
||||
},
|
||||
@@ -1158,7 +976,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
void ContextImpl::AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
ForceTSOValidRanges.Insert(ValidRanges);
|
||||
ForceTSOInstructions.merge(std::move(Instructions));
|
||||
ForceTSOInstructions.merge(Instructions);
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
|
||||
@@ -1202,5 +1020,6 @@ void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
|
||||
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
|
||||
}
|
||||
} // namespace FEXCore::Context
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -4,7 +4,6 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#include <array>
|
||||
@@ -28,10 +27,6 @@ class ContextImpl;
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
|
||||
#define STATE_PTR_IDX(STATE_TYPE, FIELD, INDEX) STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::STATE_TYPE, FIELD, INDEX)
|
||||
#define FALLBACK_HANDLER_OFFSET(INDEX, FIELD) \
|
||||
STATE.R(), \
|
||||
(ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, Pointers.FallbackHandlerPointers, INDEX) + offsetof(FEXCore::Core::FallbackABIInfo, FIELD))
|
||||
|
||||
class Dispatcher final : public Arm64Emitter {
|
||||
public:
|
||||
@@ -55,10 +50,6 @@ public:
|
||||
}
|
||||
#endif
|
||||
|
||||
uint64_t GetExitFunctionLinkerAddress() const {
|
||||
return ExitFunctionLinkerAddress;
|
||||
}
|
||||
|
||||
SignalDelegatorConfig MakeSignalDelegatorConfig() const;
|
||||
|
||||
protected:
|
||||
@@ -99,49 +90,8 @@ private:
|
||||
uint64_t LUDIVHandlerAddress {};
|
||||
uint64_t LDIVHandlerAddress {};
|
||||
|
||||
// F64 reduced-precision shared handlers
|
||||
uint64_t F64SinHandlerAddress {};
|
||||
uint64_t F64CosHandlerAddress {};
|
||||
uint64_t F64TanHandlerAddress {};
|
||||
uint64_t F64F2XM1HandlerAddress {};
|
||||
uint64_t F64ScaleHandlerAddress {};
|
||||
uint64_t F64AtanHandlerAddress {};
|
||||
uint64_t F64FYL2XHandlerAddress {};
|
||||
uint64_t F64FYL2XP1HandlerAddress {};
|
||||
uint64_t F64FPREMHandlerAddress {};
|
||||
uint64_t F64FPREM1HandlerAddress {};
|
||||
|
||||
void EmitDispatcher();
|
||||
uint64_t GenerateABICall(FallbackABI ABI);
|
||||
|
||||
// Inline softfloat conversion emitters - avoid FPCR save/restore overhead
|
||||
// These emit ARM64 code that performs the conversion using only integer ops
|
||||
void EmitI16ToExtF80();
|
||||
void EmitI32ToExtF80();
|
||||
void EmitF32ToExtF80();
|
||||
void EmitF64ToExtF80();
|
||||
|
||||
// Shared label set for the LUT-based F64 log2 path used by both FYL2X and
|
||||
// FYL2XP1. The pool is emitted once via EmitF64Log2Constants.
|
||||
struct F64Log2Constants {
|
||||
ARMEmitter::ForwardLabel One;
|
||||
ARMEmitter::ForwardLabel A0, A1, A2, A3, A4, A5, A6, A7;
|
||||
ARMEmitter::ForwardLabel Table;
|
||||
};
|
||||
|
||||
void EmitF64Sin();
|
||||
void EmitF64Cos();
|
||||
void EmitF64Tan();
|
||||
void EmitF64F2XM1();
|
||||
void EmitF64Scale();
|
||||
void EmitF64Atan();
|
||||
void EmitF64FYL2X(F64Log2Constants& C);
|
||||
void EmitF64FYL2XP1(F64Log2Constants& C);
|
||||
void EmitF64Log2Constants(F64Log2Constants& C);
|
||||
void EmitF64FPREM();
|
||||
void EmitF64FPREM1();
|
||||
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include <array>
|
||||
@@ -69,6 +70,7 @@ static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
|
||||
Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
: Thread {Thread}
|
||||
, CTX {static_cast<FEXCore::Context::ContextImpl*>(Thread->CTX)}
|
||||
, OSABI {CTX->SyscallHandler ? CTX->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN}
|
||||
, PoolObject {CTX->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
@@ -88,9 +90,9 @@ Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
}
|
||||
|
||||
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
// Check for wraparound
|
||||
if (Address + Size < Address) {
|
||||
return false;
|
||||
// Treat FEX-internal X86 callbacks as always executable
|
||||
if (EntryPoint == CTX->X86CodeGen.CallbackReturn) {
|
||||
return true;
|
||||
}
|
||||
|
||||
while (Address < ExecutableRangeBase || Address + Size > ExecutableRangeEnd) {
|
||||
@@ -114,8 +116,9 @@ bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
}
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
std::optional<uint8_t> Byte = PeekByte(0);
|
||||
if (!Byte || InstructionSize == MAX_INST_SIZE) {
|
||||
if (!Byte) {
|
||||
HitNonExecutableRange = true;
|
||||
// Pretend we read 0, the main decode loop will see HitNonExecutableRange and rollback the instruction.
|
||||
return 0;
|
||||
@@ -127,23 +130,21 @@ uint8_t Decoder::ReadByte() {
|
||||
}
|
||||
|
||||
std::optional<uint8_t> Decoder::PeekByte(uint8_t Offset) {
|
||||
uint64_t ByteAddress = reinterpret_cast<uint64_t>(InstStream.InstStream + InstructionSize + Offset);
|
||||
uint64_t ByteAddress = reinterpret_cast<uint64_t>(InstStream + InstructionSize + Offset);
|
||||
if (CheckRangeExecutable(ByteAddress, 1)) {
|
||||
return InstStream.AdjustedInstStream[InstructionSize + Offset];
|
||||
return InstStream[InstructionSize + Offset];
|
||||
} else {
|
||||
return std::nullopt;
|
||||
}
|
||||
}
|
||||
|
||||
std::pair<uint64_t, bool> Decoder::ReadData(uint8_t Size) {
|
||||
uint64_t Decoder::ReadData(uint8_t Size) {
|
||||
LOGMAN_THROW_A_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
|
||||
uint64_t Res = 0;
|
||||
uint64_t Address = reinterpret_cast<uint64_t>(InstStream.InstStream + InstructionSize);
|
||||
LastFieldReadOffset = (uint8_t)InstructionSize;
|
||||
LastFieldReadSize = Size;
|
||||
uint64_t Address = reinterpret_cast<uint64_t>(InstStream + InstructionSize);
|
||||
if (CheckRangeExecutable(Address, Size)) {
|
||||
std::memcpy(&Res, &InstStream.AdjustedInstStream[InstructionSize], Size);
|
||||
std::memcpy(&Res, &InstStream[InstructionSize], Size);
|
||||
} else {
|
||||
HitNonExecutableRange = true;
|
||||
// See PeekByte, this specific case may cause some executable memory to read as 0 but it doesn't matter as the entire instruction will be rolled back anyway.
|
||||
@@ -159,21 +160,7 @@ std::pair<uint64_t, bool> Decoder::ReadData(uint8_t Size) {
|
||||
SkipBytes(Size);
|
||||
#endif
|
||||
|
||||
if (Relocations) {
|
||||
uint32_t SectionOffset = static_cast<uint32_t>(Address - SectionMinAddress);
|
||||
if (auto It = Relocations->find(SectionOffset); It != Relocations->end()) {
|
||||
if (It->second == GuestRelocationType::Rel32 && Size == 4) {
|
||||
return {static_cast<int64_t>(static_cast<int32_t>(Res) - static_cast<int32_t>(EntryPoint)), true};
|
||||
} else if (It->second == GuestRelocationType::Rel64 && Size == 8) {
|
||||
return {static_cast<int64_t>(Res) - static_cast<int64_t>(EntryPoint), true};
|
||||
} else {
|
||||
HitBadRelocation = true;
|
||||
Res = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return {Res, false};
|
||||
return Res;
|
||||
}
|
||||
|
||||
void Decoder::DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM) {
|
||||
@@ -205,9 +192,7 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
DisplacementSize = 1;
|
||||
}
|
||||
if (DisplacementSize) {
|
||||
bool IsRelocation = false;
|
||||
std::tie(Literal, IsRelocation) = ReadData(DisplacementSize);
|
||||
LOGMAN_THROW_A_FMT(!IsRelocation, "1/2 byte relocations unsupported");
|
||||
Literal = ReadData(DisplacementSize);
|
||||
if (DisplacementSize == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
@@ -313,10 +298,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
LOGMAN_THROW_A_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
if (Displacement) {
|
||||
auto [Literal, IsRelocation] = ReadData(Displacement);
|
||||
if (IsRelocation) {
|
||||
Operand->Type = DecodedOperand::OpType::SIBRelocation;
|
||||
}
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
@@ -326,9 +308,10 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
|
||||
if (ModRM.rm == 0b101) {
|
||||
// 32bit Displacement
|
||||
auto [Literal, IsRelocation] = ReadData(4);
|
||||
Operand->Type = IsRelocation ? DecodedOperand::OpType::RIPRelativeRelocation : DecodedOperand::OpType::RIPRelative;
|
||||
Operand->Data.RIPLiteral.Value = Literal;
|
||||
const uint32_t Literal = ReadData(4);
|
||||
|
||||
Operand->Type = DecodedOperand::OpType::RIPRelative;
|
||||
Operand->Data.RIPLiteral.Value.u = Literal;
|
||||
} else {
|
||||
// Register-direct addressing
|
||||
Operand->Type = DecodedOperand::OpType::GPRDirect;
|
||||
@@ -336,18 +319,18 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
}
|
||||
} else {
|
||||
uint8_t DisplacementSize = ModRM.mod == 1 ? 1 : 4;
|
||||
auto [Literal, IsRelocation] = ReadData(DisplacementSize);
|
||||
uint32_t Literal = ReadData(DisplacementSize);
|
||||
if (DisplacementSize == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
|
||||
Operand->Type = IsRelocation ? DecodedOperand::OpType::GPRIndirectRelocation : DecodedOperand::OpType::GPRIndirect;
|
||||
Operand->Type = DecodedOperand::OpType::GPRIndirect;
|
||||
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
Operand->Data.GPRIndirect.Displacement = Literal;
|
||||
}
|
||||
}
|
||||
|
||||
Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options) {
|
||||
bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options) {
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) [[unlikely]] {
|
||||
// Dispatcher Op.
|
||||
// TODO: Move this in to `NormalOpHeader`, Dispatch tables have a bug currently where some subtables don't inherit flags correctly.
|
||||
@@ -359,16 +342,11 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
if (!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SUPPORTS_LOCK) && (DecodeInst->Flags & DecodeFlags::FLAG_LOCK)) {
|
||||
// Instruction has lock prefix but doesn't support lock.
|
||||
return DecodedBlockStatus::UNIMPLEMENTED_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P), "Group Ops "
|
||||
@@ -400,15 +378,15 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
const bool Has16BitAddressing = !BlockInfo.Is64BitMode && DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
|
||||
|
||||
if (Options.w && (Info->Flags & InstFlags::FLAGS_REX_W_0)) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
} else if (!Options.w && (Info->Flags & InstFlags::FLAGS_REX_W_1)) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Options.L && (Info->Flags & InstFlags::FLAGS_VEX_L_0)) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
} else if (!Options.L && (Info->Flags & InstFlags::FLAGS_VEX_L_1)) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
const bool UseVEXL = Options.L && !(Info->Flags & InstFlags::FLAGS_VEX_L_IGNORE);
|
||||
@@ -517,7 +495,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
|
||||
|
||||
if (CurrentDest->Data.GPR.GPR == FEXCore::X86State::REG_INVALID) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -586,7 +564,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
|
||||
const auto VEXOperand = Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_SRC_MASK;
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_NO_OPERAND && Options.vvvv) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) {
|
||||
@@ -604,11 +582,11 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM) {
|
||||
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SF_MOD_DST) {
|
||||
if (!ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest)) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
if (!ModRMOperand(DecodeInst->Dest, DecodeInst->Src[CurrentSrc], HasXMMDst, HasXMMSrc, HasMMDst, HasMMSrc, Is8BitDest, Is8BitSrc)) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
++CurrentSrc;
|
||||
@@ -639,60 +617,55 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
|
||||
}
|
||||
|
||||
if (Bytes <= 8 && Bytes > 0) {
|
||||
auto [Literal, IsRelocation] = ReadData(Bytes);
|
||||
if (IsRelocation) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::LiteralRelocation;
|
||||
DecodeInst->Src[CurrentSrc].Data.LiteralRelocation.EntrypointOffset = Literal;
|
||||
} else {
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT) ||
|
||||
(DecodeFlags::GetSizeDstFlags(DecodeInst->Flags) == DecodeFlags::SIZE_64BIT &&
|
||||
Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT64BIT)) {
|
||||
if (Bytes == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
} else if (Bytes == 2) {
|
||||
Literal = static_cast<int16_t>(Literal);
|
||||
} else {
|
||||
Literal = static_cast<int32_t>(Literal);
|
||||
}
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
|
||||
uint64_t Literal = ReadData(Bytes);
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT) || (DecodeFlags::GetSizeDstFlags(DecodeInst->Flags) == DecodeFlags::SIZE_64BIT &&
|
||||
Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT64BIT)) {
|
||||
if (Bytes == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
} else if (Bytes == 2) {
|
||||
Literal = static_cast<int16_t>(Literal);
|
||||
} else {
|
||||
Literal = static_cast<int32_t>(Literal);
|
||||
}
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.SignExtend = true;
|
||||
}
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
++CurrentSrc;
|
||||
|
||||
if (Bytes == 8) [[unlikely]] {
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = 4;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal >> 32;
|
||||
}
|
||||
|
||||
Bytes = 0;
|
||||
} else {
|
||||
// All real x86 instructions have byte sizes that are 8-bytes or less.
|
||||
// Thunk instruction has an additional 32-byte SHA256 payload that needs to be accounted for.
|
||||
InstructionSize += Bytes;
|
||||
Bytes = 0;
|
||||
}
|
||||
|
||||
if ((DecodeInst->Flags & DecodeFlags::FLAG_LOCK) && DecodeInst->Dest.IsGPR()) {
|
||||
// Instruction has lock prefix, but the destination isn't memory, this is invalid.
|
||||
return DecodedBlockStatus::UNIMPLEMENTED_INST;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
|
||||
DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return DecodedBlockStatus::SUCCESS;
|
||||
return true;
|
||||
}
|
||||
|
||||
Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op) {
|
||||
bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op) {
|
||||
DecodeInst->OPRaw = DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
|
||||
@@ -749,7 +722,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
|
||||
};
|
||||
uint8_t Field = RegToField[ModRM.reg];
|
||||
if (Field == 255) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
LocalOp = (Field << 3) | ModRM.rm;
|
||||
@@ -768,7 +741,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
|
||||
if (!VEXTable) {
|
||||
// AVX not enabled.
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
uint16_t map_select = 1;
|
||||
@@ -778,7 +751,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
|
||||
|
||||
if ((Byte1 & 0b10000000) == 0) {
|
||||
if (!BlockInfo.Is64BitMode) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
|
||||
@@ -786,28 +759,18 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
|
||||
|
||||
if (Op == 0xC5) { // Two byte VEX
|
||||
pp = Byte1 & 0b11;
|
||||
const uint8_t vvvv = ((Byte1 & 0b01111000) >> 3);
|
||||
if (!BlockInfo.Is64BitMode && vvvv <= 0b0111) {
|
||||
// Invalid on 32-bit, can't use the high registers.
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
options.vvvv = 15 - vvvv;
|
||||
options.vvvv = 15 - ((Byte1 & 0b01111000) >> 3);
|
||||
options.L = (Byte1 & 0b100) != 0;
|
||||
} else { // 0xC4 = Three byte VEX
|
||||
const uint8_t Byte2 = ReadByte();
|
||||
pp = Byte2 & 0b11;
|
||||
map_select = Byte1 & 0b11111;
|
||||
const uint8_t vvvv = ((Byte2 & 0b01111000) >> 3);
|
||||
if (!BlockInfo.Is64BitMode && vvvv <= 0b0111) {
|
||||
// Invalid on 32-bit, can't use the high registers.
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
options.vvvv = 15 - vvvv;
|
||||
options.vvvv = 15 - ((Byte2 & 0b01111000) >> 3);
|
||||
options.w = (Byte2 & 0b10000000) != 0;
|
||||
options.L = (Byte2 & 0b100) != 0;
|
||||
if ((Byte1 & 0b01000000) == 0) {
|
||||
if (!BlockInfo.Is64BitMode) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
}
|
||||
@@ -818,7 +781,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_OPTION_AVX_W;
|
||||
}
|
||||
if (!(map_select >= 1 && map_select <= 3)) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -848,14 +811,14 @@ Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X8
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(TYPE_USES_EVEX_OPS, 1);
|
||||
// EVEX unsupported
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Invalid instruction decoding type");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
InstructionSize = 0;
|
||||
LastEscapePrefix = 0;
|
||||
Instruction.fill(0);
|
||||
@@ -866,7 +829,7 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
|
||||
for (;;) {
|
||||
if (InstructionSize >= MAX_INST_SIZE) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
uint8_t Op = ReadByte();
|
||||
switch (Op) {
|
||||
@@ -875,7 +838,6 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
switch (EscapeOp) {
|
||||
case 0x0F:
|
||||
[[unlikely]] { // 3DNow!
|
||||
DecodeREXIfValid(-2);
|
||||
// 3DNow! Instruction Encoding: 0F 0F [ModRM] [SIB] [Displacement] [Opcode]
|
||||
// Decode ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
@@ -900,7 +862,6 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
break;
|
||||
}
|
||||
case 0x38: { // F38 Table!
|
||||
DecodeREXIfValid(-2);
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
@@ -926,11 +887,11 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
DecodeFlags::PopOpAddrIf(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
|
||||
}
|
||||
|
||||
return NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp);
|
||||
break;
|
||||
}
|
||||
case 0x3A: { // F3A Table!
|
||||
DecodeREXIfValid(-2);
|
||||
constexpr uint16_t PF_3A_NONE = 0;
|
||||
constexpr uint16_t PF_3A_66 = (1 << 0);
|
||||
constexpr uint16_t PF_3A_REX = (1 << 1);
|
||||
@@ -960,7 +921,6 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
bool NoOverlay = (FEXCore::X86Tables::SecondBaseOps[EscapeOp].Flags & InstFlags::FLAGS_NO_OVERLAY) != 0;
|
||||
bool NoOverlay66 = (FEXCore::X86Tables::SecondBaseOps[EscapeOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
|
||||
|
||||
DecodeREXIfValid(-2);
|
||||
if (NoOverlay) { // This section of the table ignores prefix extention
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
} else if (LastEscapePrefix == 0xF3) { // REP
|
||||
@@ -1040,9 +1000,29 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
|
||||
DecodeInst->REXIndex = InstructionSize;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
|
||||
|
||||
// Widening displacement
|
||||
if (Op & 0b1000) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_WIDENING;
|
||||
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_WIDENING_SIZE_LAST);
|
||||
}
|
||||
|
||||
// XGPR_B bit set
|
||||
if (Op & 0b0001) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
|
||||
}
|
||||
|
||||
// XGPR_X bit set
|
||||
if (Op & 0b0010) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
}
|
||||
|
||||
// XGPR_R bit set
|
||||
if (Op & 0b0100) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
|
||||
}
|
||||
} else {
|
||||
DecodeREXIfValid();
|
||||
return NormalOpHeader(Info, Op);
|
||||
}
|
||||
|
||||
@@ -1052,67 +1032,23 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
return false;
|
||||
}
|
||||
|
||||
return DecodedBlockStatus::SUCCESS;
|
||||
}
|
||||
|
||||
void Decoder::DecodeREXIfValid(int8_t ExpectedOffset) {
|
||||
LOGMAN_THROW_A_FMT(ExpectedOffset < 0, "Expecting an negative offset for the REX offset!");
|
||||
const int8_t REXIndex = InstructionSize + ExpectedOffset;
|
||||
|
||||
if (DecodeInst->REXIndex != 0 && DecodeInst->REXIndex == REXIndex) {
|
||||
const uint8_t Op = Instruction[REXIndex - 1];
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
|
||||
|
||||
// Widening displacement
|
||||
if (Op & 0b1000) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_WIDENING;
|
||||
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_WIDENING_SIZE_LAST);
|
||||
}
|
||||
|
||||
// XGPR_B bit set
|
||||
if (Op & 0b0001) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
|
||||
}
|
||||
|
||||
// XGPR_X bit set
|
||||
if (Op & 0b0010) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
}
|
||||
|
||||
// XGPR_R bit set
|
||||
if (Op & 0b0100) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
|
||||
// Will be set if DecodeInstructionImpl tries to read non-executable memory
|
||||
HitNonExecutableRange = false;
|
||||
HitBadRelocation = false;
|
||||
auto ErrorDuringDecoding = DecodeInstructionImpl(PC);
|
||||
bool ErrorDuringDecoding = !DecodeInstructionImpl(PC);
|
||||
|
||||
if (ErrorDuringDecoding != DecodedBlockStatus::SUCCESS || HitNonExecutableRange || HitBadRelocation) [[unlikely]] {
|
||||
if (ErrorDuringDecoding || HitNonExecutableRange) [[unlikely]] {
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
const auto InstSize = DecodeInst->InstSize;
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
DecodeInst->InstSize = 0;
|
||||
|
||||
// A decode error can be caused by substituting zero for an inaccessible
|
||||
// instruction byte, so the instruction fetch fault takes priority.
|
||||
if (HitNonExecutableRange) {
|
||||
return InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST : DecodedBlockStatus::NOEXEC_INST;
|
||||
}
|
||||
|
||||
if (HitBadRelocation) {
|
||||
return DecodedBlockStatus::BAD_RELOCATION;
|
||||
}
|
||||
|
||||
return ErrorDuringDecoding;
|
||||
return ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST : DecodedBlockStatus::NOEXEC_INST;
|
||||
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
@@ -1196,9 +1132,9 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// Forbid distant branches to have the cost code better match the guest code layout, avoiding massive (range-wise) code
|
||||
// blocks in highly fragmented guest code. Such branches are often not-taken branches to garbage in obfuscated code.
|
||||
constexpr uint64_t MAX_FORWARD_BRANCH_DIST = FEXCore::Utils::FEX_PAGE_SIZE * 4;
|
||||
bool ValidMultiblockMember = TargetRIP >= EntryPoint && TargetRIP < std::min(InstEnd + MAX_FORWARD_BRANCH_DIST, SectionMaxAddress);
|
||||
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < std::min(InstEnd + MAX_FORWARD_BRANCH_DIST, SymbolMaxAddress);
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _M_ARM_64EC
|
||||
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
|
||||
#endif
|
||||
|
||||
@@ -1336,13 +1272,6 @@ void Decoder::AddBranchTarget(uint64_t Target) {
|
||||
.BlockStatus = BlockIt->BlockStatus,
|
||||
};
|
||||
|
||||
if (BlockIt->DataMasks.size()) {
|
||||
auto MaskIt = std::lower_bound(BlockIt->DataMasks.begin(), BlockIt->DataMasks.end(), SplitAddr,
|
||||
[](const DataMask& Mask, uint64_t Addr) { return Mask.FieldAddress < Addr; });
|
||||
SplitBlock.DataMasks.assign(MaskIt, BlockIt->DataMasks.end());
|
||||
BlockIt->DataMasks.erase(MaskIt, BlockIt->DataMasks.end());
|
||||
}
|
||||
|
||||
BlockIt->Size = SplitOffset;
|
||||
BlockIt->NumInstructions = SplitIdx;
|
||||
|
||||
@@ -1363,165 +1292,114 @@ void Decoder::AddBranchTarget(uint64_t Target) {
|
||||
}
|
||||
}
|
||||
|
||||
const Decoder::DecodeStream Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
|
||||
const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
|
||||
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
|
||||
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
|
||||
|
||||
if (BlockInfo.Is64BitMode && CTX->HostFeatures.HostType == FEXCore::HostFeatures::HostTypeEnum::Linux && RIP >= VSyscall_Base &&
|
||||
RIP < VSyscall_End) {
|
||||
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64 && RIP >= VSyscall_Base && RIP < VSyscall_End) {
|
||||
// VSyscall
|
||||
// This doesn't exist on AArch64 and on x86_64 hosts this is emulated with faults to a region mapped with --xp permissions
|
||||
// Offset 0: vgettimeofday
|
||||
// Offset 0x400: vtime
|
||||
// Offset 0x800: vgetcpu
|
||||
uint64_t Offset = RIP - VSyscall_Base;
|
||||
return DecodeStream {
|
||||
.InstStream = _InstStream - EntryPoint + RIP,
|
||||
.AdjustedInstStream = VSyscallData + Offset,
|
||||
};
|
||||
return VSyscallData + Offset;
|
||||
}
|
||||
|
||||
return DecodeStream {
|
||||
.InstStream = _InstStream - EntryPoint + RIP,
|
||||
.AdjustedInstStream = _InstStream - EntryPoint + RIP,
|
||||
};
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
bool Decoder::CheckIfCacheable(FEXCore::Core::InternalThreadState& Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
SetupDecodeInstructionsAtEntry(&Thread, PC, MaxInst);
|
||||
DecodeLoop(InstStream);
|
||||
bool Uncacheable = HitBadRelocation;
|
||||
DelayedDisownBuffer();
|
||||
return !Uncacheable;
|
||||
}
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
VisitedBlocks.clear();
|
||||
// Reset internal state management
|
||||
DecodedSize = 0;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
void Decoder::DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block) {
|
||||
if (LastFieldReadSize < 4) {
|
||||
return;
|
||||
// Decode operating mode from thread's CS segment.
|
||||
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
|
||||
BlockInfo.Is64BitMode = CSSegment->L == 1;
|
||||
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
|
||||
|
||||
// XXX: Load symbol data
|
||||
SymbolAvailable = false;
|
||||
EntryPoint = PC;
|
||||
BlockInfo.EntryPoints = {PC};
|
||||
InstStream = _InstStream;
|
||||
|
||||
uint64_t TotalInstructions {};
|
||||
|
||||
// If we don't have symbols available then we become a bit optimistic about multiblock ranges
|
||||
if (!SymbolAvailable) {
|
||||
// If we don't have a symbol available then assume all branches are valid for multiblock
|
||||
SymbolMaxAddress = SectionMaxAddress;
|
||||
SymbolMinAddress = EntryPoint;
|
||||
}
|
||||
|
||||
FEXCore::X86Tables::DecodedOperand* LiteralToPatch = nullptr;
|
||||
DataMaskType Type;
|
||||
DecodedMinAddress = EntryPoint;
|
||||
DecodedMaxAddress = EntryPoint;
|
||||
|
||||
// mov reg,imm
|
||||
if (DecodeInst->OP >= 0xB8 && DecodeInst->OP <= 0xBF) {
|
||||
for (auto& Src : DecodeInst->Src) {
|
||||
if (Src.IsLiteral()) {
|
||||
LiteralToPatch = &Src;
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Entry is a jump target
|
||||
BlocksToDecode = {PC};
|
||||
|
||||
// we could filter to certain high values that are more likely to be pointers/etc?
|
||||
// const uint64_t Value = Lit->Data.Literal.Value;
|
||||
// if (LiteralToPatch && Value < 0x1000000ULL) {
|
||||
// LiteralToPatch = nullptr;
|
||||
// }
|
||||
Type = DataMaskType::MOV;
|
||||
uint64_t CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
|
||||
BlockInfo.CodePages = {CurrentCodePage};
|
||||
|
||||
if (MaxInst == 0) {
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
|
||||
// jmp/call branches that use a literal rip-relative offset
|
||||
// some of those may be inlined by multiblock and will be cleaned up at decode end
|
||||
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_SETS_RIP && DecodeInst->Src[0].IsLiteral()) {
|
||||
LiteralToPatch = &DecodeInst->Src[0];
|
||||
Type = DataMaskType::BRANCH;
|
||||
}
|
||||
bool EntryBlock {true};
|
||||
bool FinalInstruction {false};
|
||||
|
||||
// todo add a bunch more
|
||||
while (!FinalInstruction && !BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
VisitedBlocks.emplace(RIPToDecode);
|
||||
|
||||
if (LiteralToPatch) {
|
||||
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, Type, LastFieldReadSize});
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
|
||||
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
|
||||
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
|
||||
LiteralToPatch->Data.LiteralPatchable.Width = LastFieldReadSize;
|
||||
}
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
|
||||
|
||||
void Decoder::PruneInlinedBranchDataMasks() {
|
||||
for (auto& Block : BlockInfo.Blocks) {
|
||||
if (!Block.DataMasks.size()) {
|
||||
continue;
|
||||
NextBlockStartAddress = ~0ULL;
|
||||
if (!BlocksToDecode.empty()) {
|
||||
// We just erased the lowest, the front is then the second lowest
|
||||
NextBlockStartAddress = *BlocksToDecode.begin();
|
||||
}
|
||||
const auto& LastInst = Block.DecodedInstructions[Block.NumInstructions - 1];
|
||||
const auto& LastMask = Block.DataMasks.back();
|
||||
|
||||
if (LastMask.Type != DataMaskType::BRANCH) {
|
||||
continue;
|
||||
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = BlockSuccIt->Entry;
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(NextBlockStartAddress > RIPToDecode, "unexpected");
|
||||
|
||||
const uint64_t NextInst = LastInst.PC + LastInst.InstSize;
|
||||
if (LastMask.FieldAddress < LastInst.PC || LastMask.FieldAddress + LastMask.ValueSize > NextInst) {
|
||||
continue;
|
||||
}
|
||||
// Insert the block now so it can be looked up and split if necessary on a backward edge
|
||||
auto BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
|
||||
|
||||
if (std::ranges::binary_search(BlockInfo.Blocks, NextInst + LastInst.Src[0].Data.LiteralPatchable.Value, std::less {}, &DecodedBlocks::Entry)) {
|
||||
Block.DataMasks.pop_back();
|
||||
}
|
||||
}
|
||||
}
|
||||
BlockIt->Entry = RIPToDecode;
|
||||
BlockIt->Size = 0;
|
||||
BlockIt->IsEntryPoint = EntryBlock;
|
||||
|
||||
void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
// counter-intuitively, the masks are also needed for lookup on anon prefix decodes, not just stores
|
||||
bool WantsDataMasks = CTX->DiskCache.IsReadingDiskCache() || CTX->DiskCache.IsWritingDiskCache();
|
||||
// remove this if we ever fixup ValidateCode crc constant after relocations
|
||||
if (CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
WantsDataMasks = false;
|
||||
}
|
||||
uint64_t PCOffset = 0;
|
||||
uint64_t BlockStartOffset = DecodedSize;
|
||||
bool EraseBlock = true; // Unset once the block contains an instruction
|
||||
|
||||
while (!FinalInstruction && (Paused || !BlocksToDecode.empty())) {
|
||||
bool Pausing = false;
|
||||
fextl::vector<DecodedBlocks>::iterator BlockIt;
|
||||
if (!Paused || BlockResume == -1) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
VisitedBlocks.emplace(RIPToDecode);
|
||||
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockIt->NumInstructions = 0;
|
||||
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
|
||||
|
||||
NextBlockStartAddress = ~0ULL;
|
||||
if (!BlocksToDecode.empty()) {
|
||||
// We just erased the lowest, the front is then the second lowest
|
||||
NextBlockStartAddress = *BlocksToDecode.begin();
|
||||
}
|
||||
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = BlockSuccIt->Entry;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(NextBlockStartAddress == ~0ULL || NextBlockStartAddress > RIPToDecode, "unexpected");
|
||||
|
||||
// Insert the block now so it can be looked up and split if necessary on a backward edge
|
||||
BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
|
||||
|
||||
BlockIt->Entry = RIPToDecode;
|
||||
BlockIt->Size = 0;
|
||||
BlockIt->IsEntryPoint = EntryBlock;
|
||||
|
||||
PCOffset = 0;
|
||||
BlockStartOffset = DecodedSize;
|
||||
EraseBlock = true; // Unset once the block contains an instruction
|
||||
|
||||
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockIt->NumInstructions = 0;
|
||||
|
||||
// Do a bit of pointer math to figure out where we are in code
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
} else if (BlockResume != -1) {
|
||||
BlockIt = BlockInfo.Blocks.begin() + BlockResume;
|
||||
BlockResume = -1;
|
||||
}
|
||||
|
||||
Paused = false;
|
||||
// Do a bit of pointer math to figure out where we are in code
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
|
||||
while (1) {
|
||||
InstructionSize = 0;
|
||||
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpAddress = BlockIt->Entry + PCOffset;
|
||||
auto OpAddress = RIPToDecode + PCOffset;
|
||||
auto OpMaxAddress = OpAddress + MAX_INST_SIZE;
|
||||
|
||||
auto OpMinPage = OpAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
@@ -1543,15 +1421,7 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
BlockInfo.CodePages.insert(CurrentCodePage);
|
||||
}
|
||||
|
||||
LastFieldReadSize = 0;
|
||||
BlockIt->BlockStatus = DecodeInstruction(OpAddress);
|
||||
if (HitBadRelocation) {
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks = {*BlockIt};
|
||||
BlockInfo.EntryPoints.clear();
|
||||
BlockInfo.CodePages.clear();
|
||||
return;
|
||||
}
|
||||
uint64_t OpEndAddress = OpAddress + DecodeInst->InstSize;
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, OpAddress);
|
||||
@@ -1568,45 +1438,23 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
++BlockIt->NumInstructions;
|
||||
BlockIt->Size += DecodeInst->InstSize;
|
||||
|
||||
// if we weren't provided relocations (guest JIT), try to detect what we can
|
||||
if (WantsDataMasks && BlockIt->BlockStatus == DecodedBlockStatus::SUCCESS && BlockInfo.Is64BitMode && !Relocations) {
|
||||
DetectDataMasks(OpAddress, *BlockIt);
|
||||
}
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (BlockIt->BlockStatus != DecodedBlockStatus::SUCCESS) [[unlikely]] {
|
||||
if (!EntryBlock && BlockIt->BlockStatus != DecodedBlockStatus::BAD_RELOCATION) {
|
||||
if (!EntryBlock) {
|
||||
// In multiblock configurations, we can early terminate any non-entrypoint blocks with the expectation that this won't get hit.
|
||||
// Improves compile-times.
|
||||
// Just need to undo additions that this block decoding has caused.
|
||||
TotalInstructions -= BlockIt->NumInstructions;
|
||||
DecodedSize = BlockStartOffset;
|
||||
InstStream -= PCOffset;
|
||||
if (DecodedMaxAddress == OpEndAddress) {
|
||||
DecodedMaxAddress -= PCOffset;
|
||||
}
|
||||
EraseBlock = true;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::NOEXEC_INST ? "NoExec" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::BAD_RELOCATION ? "BadRelocation" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::UNIMPLEMENTED_INST ? "Unimplemented" :
|
||||
"PartialDecode",
|
||||
OpAddress);
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" : "NoExec", OpAddress);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
if (GuestSizePause) {
|
||||
if (GuestSizePause > DecodeInst->InstSize) {
|
||||
GuestSizePause -= DecodeInst->InstSize;
|
||||
} else {
|
||||
GuestSizePause = 0;
|
||||
Pausing = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we need to end the entire multiblock
|
||||
FinalInstruction = DecodedSize >= MaxInst || DecodedSize >= DefaultDecodedBufferSize || TotalInstructions >= MaxInst;
|
||||
if (FinalInstruction) {
|
||||
@@ -1619,12 +1467,7 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
|
||||
if (CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions)) {
|
||||
BlockIt->ForceFullSMCDetection = true;
|
||||
// todo abandon patching this for now, as the crc will fail and it will lock up redoing it over and over
|
||||
// we should fix the crc at relocation if this is important
|
||||
BlockIt->DataMasks.clear();
|
||||
}
|
||||
BlockIt->ForceFullSMCDetection = CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions);
|
||||
BranchTargetInMultiblockRange();
|
||||
}
|
||||
|
||||
@@ -1633,17 +1476,6 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
|
||||
PCOffset += DecodeInst->InstSize;
|
||||
InstStream += DecodeInst->InstSize;
|
||||
|
||||
if (Pausing) {
|
||||
Pausing = false;
|
||||
Paused = true;
|
||||
BlockResume = BlockIt - BlockInfo.Blocks.begin();
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Paused) {
|
||||
break;
|
||||
}
|
||||
|
||||
// NOTE: BlockIt is only valid here in the EraseBlock case
|
||||
@@ -1655,16 +1487,6 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
|
||||
CurrentBlockTargets.clear();
|
||||
EntryBlock = false;
|
||||
|
||||
if (Pausing && !BlocksToDecode.empty() && !FinalInstruction) {
|
||||
Paused = true;
|
||||
BlockResume = -1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Paused) {
|
||||
return;
|
||||
}
|
||||
|
||||
BlockInfo.TotalInstructionCount = TotalInstructions;
|
||||
@@ -1672,65 +1494,6 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
for (auto& Block : BlockInfo.Blocks) {
|
||||
Block.IsEntryPoint = BlockInfo.EntryPoints.contains(Block.Entry);
|
||||
}
|
||||
|
||||
// now that multiblock has settled down, remove any branch masks we put down that didn't end the block
|
||||
if (WantsDataMasks) {
|
||||
PruneInlinedBranchDataMasks();
|
||||
}
|
||||
}
|
||||
|
||||
void Decoder::SetupDecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
VisitedBlocks.clear();
|
||||
// Reset internal state management
|
||||
Paused = false;
|
||||
BlockResume = -1;
|
||||
DecodedSize = 0;
|
||||
if (MaxInst == 0) {
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
this->MaxInst = MaxInst;
|
||||
MaxCondBranchForward = 0;
|
||||
MaxCondBranchBackwards = ~0ULL;
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// Decode operating mode from thread's CS segment.
|
||||
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
|
||||
BlockInfo.Is64BitMode = CSSegment->L == 1;
|
||||
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
|
||||
|
||||
EntryPoint = PC;
|
||||
BlockInfo.EntryPoints = {PC};
|
||||
|
||||
TotalInstructions = 0;
|
||||
|
||||
SectionMinAddress = 0;
|
||||
SectionMaxAddress = ~0ULL;
|
||||
Relocations = nullptr;
|
||||
|
||||
if (CTX->GetCodeCache().IsGeneratingCache || EnableCodeCacheValidation) {
|
||||
// If generating cache, attempt to load section bounds and relocations
|
||||
if (auto SectionInfo = CTX->SyscallHandler->LookupExecutableFileSection(Thread, EntryPoint)) {
|
||||
SectionMinAddress = SectionInfo->FileStartVA;
|
||||
SectionMaxAddress = SectionInfo->EndVA;
|
||||
Relocations = &SectionInfo->FileInfo.Relocations;
|
||||
}
|
||||
}
|
||||
|
||||
DecodedMinAddress = EntryPoint;
|
||||
DecodedMaxAddress = EntryPoint;
|
||||
|
||||
// Entry is a jump target
|
||||
BlocksToDecode = {PC};
|
||||
|
||||
CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
|
||||
BlockInfo.CodePages = {CurrentCodePage};
|
||||
|
||||
EntryBlock = true;
|
||||
FinalInstruction = false;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -4,12 +4,9 @@
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
@@ -19,6 +16,9 @@
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEXCore::HLE {
|
||||
enum class SyscallOSABI;
|
||||
}
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
class Decoder final {
|
||||
@@ -27,17 +27,6 @@ public:
|
||||
SUCCESS,
|
||||
INVALID_INST,
|
||||
NOEXEC_INST,
|
||||
PARTIAL_DECODE_INST,
|
||||
BAD_RELOCATION,
|
||||
UNIMPLEMENTED_INST,
|
||||
};
|
||||
|
||||
enum class DataMaskType : uint8_t { MOV, BRANCH };
|
||||
|
||||
struct DataMask final {
|
||||
uint64_t FieldAddress;
|
||||
DataMaskType Type;
|
||||
uint8_t ValueSize;
|
||||
};
|
||||
|
||||
// New Frontend decoding
|
||||
@@ -49,7 +38,6 @@ public:
|
||||
DecodedBlockStatus BlockStatus;
|
||||
bool IsEntryPoint {};
|
||||
bool ForceFullSMCDetection {};
|
||||
fextl::vector<DataMask> DataMasks;
|
||||
};
|
||||
|
||||
struct DecodedBlockInformation final {
|
||||
@@ -61,10 +49,7 @@ public:
|
||||
};
|
||||
|
||||
Decoder(FEXCore::Core::InternalThreadState* Thread);
|
||||
bool CheckIfCacheable(FEXCore::Core::InternalThreadState&, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
void SetupDecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t PC, uint64_t MaxInst);
|
||||
void DecodeLoop(const uint8_t* InstStream, uint64_t GuestPause = 0);
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
const DecodedBlockInformation* GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
@@ -73,6 +58,9 @@ public:
|
||||
uint64_t DecodedMinAddress {};
|
||||
uint64_t DecodedMaxAddress {~0ULL};
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) {
|
||||
SectionMaxAddress = v;
|
||||
}
|
||||
void SetExternalBranches(fextl::set<uint64_t>* v) {
|
||||
ExternalBranches = v;
|
||||
}
|
||||
@@ -81,10 +69,6 @@ public:
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
void ValidateDisownedOrFree() const {
|
||||
PoolObject.ValidateDisownedOrFree();
|
||||
}
|
||||
|
||||
void ResetExecutableRangeCache() {
|
||||
ExecutableRangeBase = ExecutableRangeEnd = 0;
|
||||
}
|
||||
@@ -100,10 +84,9 @@ private:
|
||||
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
const FEXCore::HLE::SyscallOSABI OSABI {};
|
||||
|
||||
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
|
||||
|
||||
DecodedBlockStatus DecodeInstructionImpl(uint64_t PC);
|
||||
bool DecodeInstructionImpl(uint64_t PC);
|
||||
DecodedBlockStatus DecodeInstruction(uint64_t PC);
|
||||
|
||||
void BranchTargetInMultiblockRange();
|
||||
@@ -112,86 +95,47 @@ private:
|
||||
|
||||
void AddBranchTarget(uint64_t Target);
|
||||
|
||||
void DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block);
|
||||
void PruneInlinedBranchDataMasks();
|
||||
|
||||
bool CheckRangeExecutable(uint64_t Address, uint64_t Size);
|
||||
|
||||
uint8_t ReadByte();
|
||||
std::optional<uint8_t> PeekByte(uint8_t Offset);
|
||||
std::pair<uint64_t, bool> ReadData(uint8_t Size);
|
||||
|
||||
uint64_t ReadData(uint8_t Size);
|
||||
void SkipBytes(uint8_t Size) {
|
||||
InstructionSize += Size;
|
||||
}
|
||||
|
||||
DecodedBlockStatus NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
|
||||
DecodedBlockStatus NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
|
||||
|
||||
void DecodeREXIfValid(int8_t ExpectedOffset = -1);
|
||||
bool NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
|
||||
bool NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
|
||||
|
||||
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
|
||||
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
|
||||
Utils::PoolBufferWithTimedRetirement<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
uint64_t TotalInstructions {};
|
||||
uint64_t CurrentCodePage {};
|
||||
bool EntryBlock {};
|
||||
bool FinalInstruction {};
|
||||
uint64_t MaxInst {};
|
||||
bool Paused {};
|
||||
int64_t BlockResume = -1;
|
||||
uint64_t PCOffset {};
|
||||
uint64_t BlockStartOffset {};
|
||||
bool EraseBlock {};
|
||||
|
||||
uint8_t LastFieldReadOffset;
|
||||
uint8_t LastFieldReadSize;
|
||||
|
||||
uint64_t ExecutableRangeBase {};
|
||||
uint64_t ExecutableRangeEnd {};
|
||||
bool ExecutableRangeWritable {};
|
||||
bool HitNonExecutableRange {};
|
||||
bool HitBadRelocation {};
|
||||
|
||||
struct DecodeStream {
|
||||
// Original instruction stream RIP location.
|
||||
const uint8_t* InstStream;
|
||||
|
||||
// Adjusted location for FEX actually decodes from.
|
||||
const uint8_t* AdjustedInstStream;
|
||||
|
||||
DecodeStream& operator-=(size_t offset) noexcept {
|
||||
InstStream -= offset;
|
||||
AdjustedInstStream -= offset;
|
||||
return *this;
|
||||
}
|
||||
|
||||
DecodeStream& operator+=(size_t offset) noexcept {
|
||||
InstStream += offset;
|
||||
AdjustedInstStream += offset;
|
||||
return *this;
|
||||
}
|
||||
};
|
||||
|
||||
DecodeStream InstStream;
|
||||
const uint8_t* InstStream {};
|
||||
IR::OpSize GetGPROpSize() const {
|
||||
return BlockInfo.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
|
||||
}
|
||||
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize {};
|
||||
// Contains the full decoded instruction, unless it is a `Thunk` instruction.
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
uint8_t LastEscapePrefix {};
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
// This is for multiblock data tracking
|
||||
bool SymbolAvailable {false};
|
||||
uint64_t EntryPoint {};
|
||||
uint64_t MaxCondBranchForward {};
|
||||
uint64_t MaxCondBranchBackwards {~0ULL};
|
||||
uint64_t SymbolMaxAddress {};
|
||||
uint64_t SymbolMinAddress {~0ULL};
|
||||
uint64_t SectionMaxAddress {~0ULL};
|
||||
uint64_t SectionMinAddress {};
|
||||
uint64_t NextBlockStartAddress {~0ULL};
|
||||
|
||||
DecodedBlockInformation BlockInfo;
|
||||
@@ -200,8 +144,6 @@ private:
|
||||
fextl::set<uint64_t> VisitedBlocks;
|
||||
fextl::set<uint64_t>* ExternalBranches {nullptr};
|
||||
|
||||
const fextl::robin_map<uint32_t, GuestRelocationType>* Relocations {nullptr};
|
||||
|
||||
// ModRM rm decoding
|
||||
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
|
||||
void DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
|
||||
@@ -217,6 +159,6 @@ private:
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_TABLE_SIZE>* VEXTable {};
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_GROUP_TABLE_SIZE>* VEXTableGroup {};
|
||||
|
||||
const DecodeStream AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
};
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -302,16 +302,6 @@ struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2XP1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
ScopedSoftFloatState State {FCW, Frame, true};
|
||||
const X80SoftFloat One {&State.State, 1.0};
|
||||
return X80SoftFloat::FYL2X(&State.State, X80SoftFloat::FADD(&State.State, Src1, One), Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
@@ -427,14 +417,6 @@ struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2XP1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return src2 * log2(1.0 + src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
@@ -453,12 +435,12 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
X80SoftFloat Src1 = Src1q;
|
||||
ScopedSoftFloatState State {FCW, Frame};
|
||||
bool Negative = Src1.Top.Sign;
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
Src1 = X80SoftFloat::FRNDINT(&State.State, Src1);
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Top.Sign = 0;
|
||||
Src1.Sign = 0;
|
||||
|
||||
uint64_t Tmp = Src1.ToI64(&State.State);
|
||||
X80SoftFloat Rv;
|
||||
@@ -521,7 +503,7 @@ struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
X80SoftFloat Tmp;
|
||||
|
||||
Tmp = BCD;
|
||||
Tmp.Top.Sign = Negative;
|
||||
Tmp.Sign = Negative;
|
||||
return Tmp;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -10,6 +10,11 @@
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_UNKNOWN, HandlerIndex};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint64_t* ABIHandlers) {
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = {ABIHandlers[FABI_F80_I16_F32_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4)};
|
||||
@@ -67,8 +72,6 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80DIV>::handle)};
|
||||
Info[Core::OPINDEX_F80FYL2X] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2X>::handle)};
|
||||
Info[Core::OPINDEX_F80FYL2XP1] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2XP1>::handle)};
|
||||
Info[Core::OPINDEX_F80ATAN] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80ATAN>::handle)};
|
||||
Info[Core::OPINDEX_F80FPREM1] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
|
||||
@@ -94,8 +97,6 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle)};
|
||||
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle)};
|
||||
Info[Core::OPINDEX_F64FYL2XP1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2XP1>::handle)};
|
||||
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle)};
|
||||
|
||||
@@ -211,6 +212,12 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
@@ -247,7 +254,6 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
COMMON_BINARY_X87_OP(MUL)
|
||||
COMMON_BINARY_X87_OP(DIV)
|
||||
COMMON_BINARY_X87_OP(FYL2X)
|
||||
COMMON_BINARY_X87_OP(FYL2XP1)
|
||||
COMMON_BINARY_X87_OP(ATAN)
|
||||
COMMON_BINARY_X87_OP(FPREM1)
|
||||
COMMON_BINARY_X87_OP(FPREM)
|
||||
@@ -262,7 +268,6 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
|
||||
// Double Precision Binary
|
||||
COMMON_BINARY_F64_OP(FYL2X)
|
||||
COMMON_BINARY_F64_OP(FYL2XP1)
|
||||
COMMON_BINARY_F64_OP(ATAN)
|
||||
COMMON_BINARY_F64_OP(FPREM1)
|
||||
COMMON_BINARY_F64_OP(FPREM)
|
||||
|
||||
@@ -2,14 +2,14 @@
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
#ifdef _M_ARM_64
|
||||
#include <arm_neon.h>
|
||||
#endif
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
#ifdef _M_ARM_64
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
|
||||
@@ -13,6 +13,9 @@ $end_info$
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
|
||||
#define DEF_BINOP_WITH_CONSTANT(FEXOp, VarOp, ConstOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
@@ -40,33 +43,21 @@ DEF_BINOP_WITH_CONSTANT(Ror, rorv, ror)
|
||||
DEF_OP(Constant) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
const auto PadType = [Pad = Op->Pad]() {
|
||||
switch (Pad) {
|
||||
case IR::ConstPad::NoPad: return CPU::Arm64Emitter::PadType::NOPAD;
|
||||
case IR::ConstPad::DoPad: return CPU::Arm64Emitter::PadType::DOPAD;
|
||||
default: return CPU::Arm64Emitter::PadType::AUTOPAD;
|
||||
}
|
||||
}();
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Op->Constant, PadType, Op->MaxBytes);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Op->Constant);
|
||||
}
|
||||
|
||||
DEF_OP(EntrypointOffset) {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
|
||||
auto Constant = Entry + Op->Offset;
|
||||
auto Dst = GetReg(Node);
|
||||
uint64_t Mask = ~0ULL;
|
||||
const auto OpSize = IROp->Size;
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
InsertGuestRIPMove(GetReg(Node), Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(PatchableGuestData) {
|
||||
auto Op = IROp->C<IR::IROp_PatchableGuestData>();
|
||||
InsertGuestPatchableDataMove(GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
@@ -276,7 +267,7 @@ DEF_OP(CmpPairZ) {
|
||||
|
||||
// Restore NzCV
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
rmif(TMP1, 28, 0xb /* NzCV */);
|
||||
rmif(TMP1, 0, 0xb /* NzCV */);
|
||||
} else {
|
||||
cset(ARMEmitter::Size::i32Bit, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP1, TMP2, 30 /* lsb: Z */, 1);
|
||||
@@ -381,7 +372,7 @@ DEF_OP(CondSubNZCV) {
|
||||
DEF_OP(Neg) {
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
|
||||
if (Op->Cond == IR::CondClass::AL) {
|
||||
if (Op->Cond == FEXCore::IR::COND_AL) {
|
||||
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src));
|
||||
} else {
|
||||
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src), MapCC(Op->Cond));
|
||||
@@ -423,8 +414,8 @@ DEF_OP(MulH) {
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
sxtw(TMP1, Src1.W());
|
||||
sxtw(TMP2, Src2.W());
|
||||
mul(ARMEmitter::Size::i64Bit, Dst, TMP1, TMP2);
|
||||
ubfx(ARMEmitter::Size::i64Bit, Dst, Dst, 32, 32);
|
||||
mul(ARMEmitter::Size::i32Bit, Dst, TMP1, TMP2);
|
||||
ubfx(ARMEmitter::Size::i32Bit, Dst, Dst, 32, 32);
|
||||
} else {
|
||||
smulh(Dst.X(), Src1.X(), Src2.X());
|
||||
}
|
||||
@@ -525,7 +516,7 @@ DEF_OP(AndWithFlags) {
|
||||
}
|
||||
|
||||
DEF_OP(AndShift) {
|
||||
auto Op = IROp->C<IR::IROp_AndShift>();
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
|
||||
and_(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
@@ -723,7 +714,7 @@ DEF_OP(Extr) {
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
auto Op = IROp->C<IR::IROp_PDep>();
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Dest = GetReg(Node);
|
||||
@@ -776,7 +767,7 @@ DEF_OP(PDep) {
|
||||
// Now, they're copied, so we can start setting Dest (even if it overlaps with
|
||||
// one of them). Handle early exit case
|
||||
mov(EmitSize, Dest, 0);
|
||||
(void)cbz(EmitSize, Mask, &Done);
|
||||
(void)cbz(EmitSize, OrigMask, &Done);
|
||||
|
||||
// Setup for first iteration
|
||||
neg(EmitSize, T0, Mask);
|
||||
@@ -926,7 +917,7 @@ DEF_OP(Div) {
|
||||
mov(EmitSize, TMP2, Lower);
|
||||
mov(EmitSize, TMP3, Divisor);
|
||||
|
||||
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.LDIVHandler));
|
||||
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP4);
|
||||
@@ -1009,7 +1000,7 @@ DEF_OP(UDiv) {
|
||||
mov(EmitSize, TMP2, Lower);
|
||||
mov(EmitSize, TMP3, Divisor);
|
||||
|
||||
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.LUDIVHandler));
|
||||
ldr(TMP4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP4);
|
||||
@@ -1199,19 +1190,6 @@ DEF_OP(Rev) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Rbit) {
|
||||
auto Op = IROp->C<IR::IROp_Rbit>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
rbit(EmitSize, Dst, Src);
|
||||
}
|
||||
|
||||
DEF_OP(Bfi) {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
@@ -1295,16 +1273,6 @@ DEF_OP(Sbfe) {
|
||||
sbfx(ConvertSize(IROp), Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
|
||||
DEF_OP(MaskGenerateFromBitWidth) {
|
||||
auto Op = IROp->C<IR::IROp_MaskGenerateFromBitWidth>();
|
||||
auto BitWidth = GetReg(Op->BitWidth);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1);
|
||||
cmp(ARMEmitter::Size::i64Bit, BitWidth, 0);
|
||||
lslv(ARMEmitter::Size::i64Bit, TMP2, TMP1, BitWidth);
|
||||
csinv(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
}
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -11,33 +11,36 @@ $end_info$
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl& CTX, FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
|
||||
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return CTX.Dispatcher->GetExitFunctionLinkerAddress();
|
||||
|
||||
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op)); break;
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum) {
|
||||
Relocation MoveABI {};
|
||||
MoveABI.NamedThunkMove.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
// Pointers are required to fit within 48-bit VA space.
|
||||
// TODO: Force 6-byte `MaxSize`, with zext extension to 64-bit. Current code not smart enough to handle negatives.
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, FEXCore::CPU::Arm64Emitter::PadType::AUTOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, false);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(*CTX, Op);
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
|
||||
NamedSymbolLiteralPair Lit {
|
||||
Arm64JITCore::NamedSymbolLiteralPair Lit {
|
||||
.Lit = Pointer,
|
||||
.MoveABI =
|
||||
{
|
||||
@@ -45,119 +48,92 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
|
||||
{
|
||||
.Header =
|
||||
{
|
||||
.Offset = 0, // Set by PlaceNamedSymbolLiteral
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit) {
|
||||
switch (Lit.MoveABI.Header.Type) {
|
||||
case RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL:
|
||||
case RelocationTypes::RELOC_GUEST_RIP_LITERAL:
|
||||
case RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
|
||||
Lit.MoveABI.Header.Offset = GetCursorOffset();
|
||||
break;
|
||||
}
|
||||
|
||||
default: ERROR_AND_DIE_FMT("Unknown relocation type for {}", __FUNCTION__);
|
||||
}
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
|
||||
BindOrRestart(&Lit.Loc);
|
||||
dc64(Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
auto Arm64JITCore::InsertGuestRIPLiteral(uint64_t GuestRIP) -> NamedSymbolLiteralPair {
|
||||
return {
|
||||
.Lit = GuestRIP,
|
||||
.MoveABI =
|
||||
{
|
||||
.GuestRIP = {.Header =
|
||||
{
|
||||
.Offset = 0, // Set by PlaceNamedSymbolLiteral
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL,
|
||||
},
|
||||
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
|
||||
.GuestRIP = GuestRIP},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI {};
|
||||
MoveABI.GuestRIP.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE};
|
||||
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
|
||||
MoveABI.GuestRIP.GuestRIP = Constant;
|
||||
MoveABI.GuestRIP.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
// Pointers are required to fit within 48-bit VA space.
|
||||
// TODO: Force 6-byte `MaxSize`, with sign extension to 64-bit. Current code not smart enough to handle negatives.
|
||||
// 48-bit sign extension works because x86-64 guests only receive 47-bit VA space, with 48-bit being reserved for kernel.
|
||||
// Additional quirk, "canonical" 48-bit pointers on x86-64, sign extend the 48-bit as well (Which is why kernel pointers are negative).
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, FEXCore::CPU::Arm64Emitter::PadType::AUTOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, false);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
auto Arm64JITCore::InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize) -> NamedSymbolLiteralPair {
|
||||
return {
|
||||
.Lit = GuestRIP,
|
||||
.MoveABI =
|
||||
{
|
||||
.GuestPatchableData = {.Header =
|
||||
{
|
||||
.Offset = 0, // Set by PlaceNamedSymbolLiteral
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL,
|
||||
},
|
||||
.RegisterIndex = 0, // unused
|
||||
.ValueSize = ValueSize,
|
||||
// NOTE: Cache serialization will subtract the unit entry address later
|
||||
.SiteAddress = SiteAddress},
|
||||
},
|
||||
};
|
||||
}
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation> Relocations) {
|
||||
const auto OrigBase = GetBufferBase();
|
||||
const auto OrigSize = GetBufferSize();
|
||||
const auto OrigOffset = GetCursorOffset();
|
||||
|
||||
void Arm64JITCore::InsertGuestPatchableDataMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
Relocation MoveABI = Relocation::Default();
|
||||
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE};
|
||||
MoveABI.GuestPatchableData.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestPatchableData.ValueSize = ValueSize;
|
||||
MoveABI.GuestPatchableData.SiteAddress = SiteAddress;
|
||||
SetBuffer(reinterpret_cast<std::uint8_t*>(Code.data()), Code.size_bytes());
|
||||
for (auto& Reloc : Relocations) {
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc.NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
SetCursorOffset(Reloc.NamedSymbolLiteral.Offset);
|
||||
|
||||
// this might get patched on disk cache load
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Value, FEXCore::CPU::Arm64Emitter::PadType::DOPAD);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
// Generate a literal so we can place it
|
||||
dc64(Pointer);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestPatchableRIPMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
Relocation MoveABI = Relocation::Default();
|
||||
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE};
|
||||
MoveABI.GuestPatchableData.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestPatchableData.ValueSize = ValueSize;
|
||||
MoveABI.GuestPatchableData.SiteAddress = SiteAddress;
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(Reloc.NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
return false;
|
||||
}
|
||||
|
||||
// this might get patched on disk cache load
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Value, FEXCore::CPU::Arm64Emitter::PadType::DOPAD);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations(uint64_t GuestBaseAddress) {
|
||||
// Rebase relocations to library base address
|
||||
for (auto& Relocation : Relocations) {
|
||||
switch (Relocation.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE:
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
Relocation.GuestRIP.GuestRIP -= GuestBaseAddress;
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(Reloc.GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
default:;
|
||||
}
|
||||
}
|
||||
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations() {
|
||||
return std::move(Relocations);
|
||||
}
|
||||
|
||||
|
||||
@@ -138,6 +138,26 @@ DEF_OP(CAS) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
steorl(SubEmitSize, Src, MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
eor(EmitSize, TMP2, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
|
||||
(void)cbnz(EmitSize, TMP2, &LoopTop);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -329,7 +349,7 @@ DEF_OP(TelemetrySetValue) {
|
||||
auto Op = IROp->C<IR::IROp_TelemetrySetValue>();
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
ldr(TMP2, STATE_PTR_IDX(CpuStateFrame, Pointers.TelemetryValueAddresses, Op->TelemetryValueIndex));
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.TelemetryValueAddresses[Op->TelemetryValueIndex]));
|
||||
|
||||
// Cortex fuses cmp+cset.
|
||||
cmp(ARMEmitter::Size::i32Bit, Src, 0);
|
||||
@@ -342,8 +362,8 @@ DEF_OP(TelemetrySetValue) {
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
|
||||
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
|
||||
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP4, TMP3, TMP2);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP4, &LoopTop);
|
||||
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -55,40 +55,13 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
uint64_t NewRIP;
|
||||
|
||||
if constexpr (Context::BLOCK_DEBUGGING) {
|
||||
// Skip block linking when BLOCK_DEBUGGING as it adds overhead and is unncessary.
|
||||
// This is a debug only feature and doesn't need caching help.
|
||||
bool IsInlineRIP = IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP);
|
||||
ARMEmitter::ForwardLabel l_ExitLink;
|
||||
if (IsInlineRIP) {
|
||||
ldr(TMP1, &l_ExitLink);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
} else {
|
||||
auto RipReg = GetReg(Op->NewRIP);
|
||||
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
}
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
|
||||
br(TMP2);
|
||||
|
||||
if (IsInlineRIP) {
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(NewRIP);
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _M_ARM_64EC
|
||||
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
|
||||
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
if (Op->PatchSiteAddress) {
|
||||
InsertGuestPatchableRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP, Op->PatchSiteAddress, Op->PatchSiteSize);
|
||||
} else {
|
||||
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
}
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionEC));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
} else {
|
||||
#endif
|
||||
@@ -177,17 +150,16 @@ DEF_OP(ExitFunction) {
|
||||
ARMEmitter::ForwardLabel TFUnset;
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
|
||||
// todo do we need to account for cache patching here?
|
||||
InsertGuestRIPMove(TMP1, NewRIP);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, NewRIP);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
blr(TMP2);
|
||||
(void)Bind(&TFUnset);
|
||||
}
|
||||
|
||||
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call, Op->PatchSiteAddress, Op->PatchSiteSize);
|
||||
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
|
||||
(void)Bind(&l_CallReturn);
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _M_ARM_64EC
|
||||
}
|
||||
#endif
|
||||
} else {
|
||||
@@ -202,19 +174,21 @@ DEF_OP(ExitFunction) {
|
||||
}
|
||||
|
||||
// L1 Cache
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// L1Mask is pre-shifted.
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
add(TMP1, TMP1, TMP2);
|
||||
// arithmetic. ubfiz+add is marginally faster on Firestorm than
|
||||
// and+add(shift). Same performance on Cortex.
|
||||
static_assert(LookupCache::L1_ENTRIES_MASK == ((1u << 20) - 1));
|
||||
ubfiz(ARMEmitter::Size::i64Bit, TMP4, RipReg, 4, 20);
|
||||
add(TMP1, TMP1, TMP4);
|
||||
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
|
||||
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
sub(TMP1, TMP1, RipReg.X());
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
|
||||
(void)Bind(&SkipFullLookup);
|
||||
@@ -261,16 +235,16 @@ DEF_OP(CondJump) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
|
||||
if (Op->Cond == IR::CondClass::EQ) {
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond == IR::CondClass::NEQ) {
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbnz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond == IR::CondClass::TSTZ) {
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond == IR::CondClass::TSTNZ) {
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbnz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else {
|
||||
@@ -282,19 +256,24 @@ DEF_OP(CondJump) {
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
// Arguments are passed as follows:
|
||||
// X0: SyscallHandler
|
||||
// X1: ThreadState
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) == FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
|
||||
// Need to spill all caller saved registers still
|
||||
GPRSpillMask = CALLER_GPR_MASK;
|
||||
FPRSpillMask = CALLER_FPR_MASK;
|
||||
}
|
||||
|
||||
SpillStaticRegs(TMP1, {
|
||||
.GPRSpillMask = GPRSpillMask,
|
||||
.FPRSpillMask = FPRSpillMask,
|
||||
});
|
||||
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -303,30 +282,141 @@ DEF_OP(Syscall) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerFunc));
|
||||
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
str(GetReg(Op->Header.Args[i]).X(), ARMEmitter::Reg::rsp, i * 8);
|
||||
}
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
|
||||
|
||||
// SP supporting move
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
// Syscall result is in any static register that the frontend desired.
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r1,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
.GPRFillMask = GPRSpillMask,
|
||||
.FPRFillMask = FPRSpillMask,
|
||||
});
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
|
||||
|
||||
PopDynamicRegs();
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegs();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(InlineSyscall) {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
// Arguments are passed as follows:
|
||||
// X8: SyscallNumber - RA INTERSECT
|
||||
// X0: Arg0 & Return
|
||||
// X1: Arg1
|
||||
// X2: Arg2
|
||||
// X3: Arg3
|
||||
// X4: Arg4 - RA INTERSECT
|
||||
// X5: Arg5 - RA INTERSECT
|
||||
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
|
||||
|
||||
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
|
||||
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS - 1> RegArgs = {
|
||||
{ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5}};
|
||||
|
||||
bool Intersects {};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
|
||||
|
||||
SpillMask |= (1U << Reg.Idx());
|
||||
Intersects = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Now that we have claimed to be a syscall we can set up the arguments
|
||||
const auto EmitSize = CTX->Config.Is64BitMode() ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSubSize = CTX->Config.Is64BitMode() ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
|
||||
if (Intersects) {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (SpillMask & (1U << Reg.Idx())) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RDX, and RSP. Which have just been spilled
|
||||
// Just load back from the context.
|
||||
auto Correlation = GetX86RegRelationToARMReg(Reg);
|
||||
LOGMAN_THROW_A_FMT(Correlation != X86State::REG_INVALID, "Invalid register mapping");
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[Correlation]));
|
||||
} else {
|
||||
mov(EmitSize, RegArgs[i].R(), Reg);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i]));
|
||||
}
|
||||
}
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, Op->HostSyscallNumber);
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask, ~0U, ARMEmitter::Reg::r8, ARMEmitter::Reg::r1);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Result is now in x0
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
@@ -335,16 +425,14 @@ DEF_OP(Thunk) {
|
||||
// X0: CTX
|
||||
// X1: Args (from guest stack)
|
||||
|
||||
// spill to ctx before ra64 spill
|
||||
SpillStaticRegs(TMP1, {
|
||||
.NZCV = false,
|
||||
});
|
||||
SpillStaticRegs(TMP1); // spill to ctx before ra64 spill
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr));
|
||||
|
||||
InsertNamedThunkRelocation(ARMEmitter::Reg::r2, Op->ThunkNameHash);
|
||||
auto thunkFn = static_cast<Context::ContextImpl*>(ThreadState->CTX)->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
@@ -353,69 +441,58 @@ DEF_OP(Thunk) {
|
||||
|
||||
PopDynamicRegs();
|
||||
|
||||
// load from ctx after ra64 refill
|
||||
FillStaticRegs({
|
||||
.NZCV = false,
|
||||
});
|
||||
FillStaticRegs(); // load from ctx after ra64 refill
|
||||
}
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
auto Base = GetReg(Op->Address).X();
|
||||
auto OldCode = Op->CodeOriginal.data();
|
||||
auto Base = GetReg(Op->Header.Args[0]).X();
|
||||
int len = Op->CodeLength;
|
||||
int Offset = 0;
|
||||
ARMEmitter::ForwardLabel Fail;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto CRC32Reg = GetReg(Op->crc);
|
||||
|
||||
// Changes to TMP1
|
||||
auto WorkingReg = ARMEmitter::XReg::zr;
|
||||
auto BaseReg = TMP2;
|
||||
auto TmpDataReg = TMP3;
|
||||
mov(ARMEmitter::Size::i64Bit, BaseReg, Base);
|
||||
auto EmitCheck = [&](size_t Size, auto&& LoadData) {
|
||||
while (len >= Size) {
|
||||
LoadData();
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
|
||||
len -= Size;
|
||||
Offset += Size;
|
||||
}
|
||||
};
|
||||
|
||||
while (len >= 8) {
|
||||
ldr<ARMEmitter::IndexType::POST>(TmpDataReg, BaseReg, 8);
|
||||
crc32x(TMP1, WorkingReg, TmpDataReg);
|
||||
len -= 8;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
EmitCheck(8, [&]() {
|
||||
ldr(TMP1, Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
while (len >= 4) {
|
||||
ldr<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 4);
|
||||
crc32w(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
|
||||
len -= 4;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
EmitCheck(4, [&]() {
|
||||
ldr(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
while (len >= 2) {
|
||||
ldrh<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 2);
|
||||
crc32h(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
|
||||
len -= 2;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
EmitCheck(2, [&]() {
|
||||
ldrh(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
while (len >= 1) {
|
||||
ldrb<ARMEmitter::IndexType::POST>(TmpDataReg.W(), BaseReg, 1);
|
||||
crc32b(TMP1.W(), WorkingReg.W(), TmpDataReg.W());
|
||||
len -= 1;
|
||||
WorkingReg = TMP1;
|
||||
}
|
||||
|
||||
sub(ARMEmitter::Size::i32Bit, Dst, TMP1, CRC32Reg);
|
||||
EmitCheck(1, [&]() {
|
||||
ldrb(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
ARMEmitter::ForwardLabel End;
|
||||
cbz_OrRestart(ARMEmitter::Size::i32Bit, Dst, &End);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
|
||||
b_OrRestart(&End);
|
||||
BindOrRestart(&Fail);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
|
||||
BindOrRestart(&End);
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
auto Op = IROp->C<IR::IROp_ThreadRemoveCodeEntry>();
|
||||
|
||||
// Move the entry to ABI before saving state.
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->Entry));
|
||||
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
@@ -424,7 +501,9 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// X1: RIP
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadRemoveCodeEntryFromJIT));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
@@ -449,8 +528,8 @@ DEF_OP(CPUID) {
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
// x2 = CPUID Leaf
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.CPUIDFunction));
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction));
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, TMP2);
|
||||
@@ -490,8 +569,8 @@ DEF_OP(XGetBV) {
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = XCR Function
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.XCRFunction));
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
|
||||
@@ -292,6 +292,8 @@ DEF_OP(Vector_FToS) {
|
||||
frinti(SubEmitSize, Dst.Z(), Mask.Merging(), Vector.Z());
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Dst.Z(), SubEmitSize);
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
frinti(SubEmitSize, Dst.D(), Vector.D());
|
||||
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
|
||||
@@ -421,11 +423,11 @@ DEF_OP(Vector_FToI) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Op->Round) {
|
||||
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
} else {
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
@@ -447,21 +449,21 @@ DEF_OP(Vector_FToI) {
|
||||
}
|
||||
|
||||
switch (Op->Round) {
|
||||
case IR::RoundMode::Nearest: ROUNDING_FN(frintn); break;
|
||||
case IR::RoundMode::NegInfinity: ROUNDING_FN(frintm); break;
|
||||
case IR::RoundMode::PosInfinity: ROUNDING_FN(frintp); break;
|
||||
case IR::RoundMode::TowardsZero: ROUNDING_FN(frintz); break;
|
||||
case IR::RoundMode::Host: ROUNDING_FN(frinti); break;
|
||||
case IR::Round_Nearest.Val: ROUNDING_FN(frintn); break;
|
||||
case IR::Round_Negative_Infinity.Val: ROUNDING_FN(frintm); break;
|
||||
case IR::Round_Positive_Infinity.Val: ROUNDING_FN(frintp); break;
|
||||
case IR::Round_Towards_Zero.Val: ROUNDING_FN(frintz); break;
|
||||
case IR::Round_Host.Val: ROUNDING_FN(frinti); break;
|
||||
}
|
||||
|
||||
#undef ROUNDING_FN
|
||||
} else {
|
||||
switch (Op->Round) {
|
||||
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -537,11 +539,11 @@ DEF_OP(Vector_F64ToI32) {
|
||||
// Then convert to integers using fcvtzs.
|
||||
auto CVTReg = Dst.Z();
|
||||
switch (Round) {
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::TowardsZero: CVTReg = Vector.Z(); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Towards_Zero.Val: CVTReg = Vector.Z(); break;
|
||||
case IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
|
||||
fcvtzs(Dst.Z(), ARMEmitter::SubRegSize::i32Bit, Mask, CVTReg, ARMEmitter::SubRegSize::i64Bit);
|
||||
@@ -557,30 +559,26 @@ DEF_OP(Vector_F64ToI32) {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// This has a known precision issue that isn't easily resolvable without throwing away performance.
|
||||
// Doing the conversion in multi-stage steps has an issue that you can lose precision in the f32->i32 step if your source was f64.
|
||||
// To get around this with ASIMD FEX needs to use fcvtzs (Scalar, Integer, to GPR) for each F64 to be directly converted to i32.
|
||||
// This is a very costly transform that the SVE path doesn't need to do since it supports f64->i32 directly.
|
||||
// If this precision issue is necessary then we can add an option for it in the future.
|
||||
|
||||
///< Round float to integral depending on rounding mode.
|
||||
///< skip TowardsZero as fcvtzs below already truncates toward zero on its own
|
||||
auto CVTReg = Dst.Q();
|
||||
switch (Round) {
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: CVTReg = Vector.Q(); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
|
||||
///< Convert f64 directly to i64
|
||||
fcvtzs(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), CVTReg);
|
||||
// Now narrow from f64 to f32.
|
||||
fcvtn(ARMEmitter::SubRegSize::i32Bit, Dst.Q(), Dst.Q());
|
||||
|
||||
///< Saturating narrow i64 -> i32
|
||||
///
|
||||
///< The caller(Vector_CVT_Float_To_Int32Impl) only fixes up positive overflow:
|
||||
///< it tests MaxF > Src (MaxF = 2^31) and swaps in CVTMAX_I32 (0x80000000) where
|
||||
///< the test fails.
|
||||
///
|
||||
///< Sources below INT32_MIN are handled by sqxtn:
|
||||
///< ARM saturates to INT32_MIN, which is 0x80000000 the same value as
|
||||
///< x86's integer-indefinite value.
|
||||
sqxtn(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Dst.D());
|
||||
///< Convert the two F32 integrals to real integers.
|
||||
fcvtzs(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Dst.D());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -324,62 +324,24 @@ DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: {
|
||||
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), Src2.Z());
|
||||
break;
|
||||
}
|
||||
case 0b00000001: {
|
||||
trn2(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), Src1.Z(), Src1.Z());
|
||||
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), VTMP1.Z(), Src2.Z());
|
||||
break;
|
||||
}
|
||||
case 0b00010000:
|
||||
trn2(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), Src2.Z(), Src2.Z());
|
||||
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), VTMP1.Z());
|
||||
break;
|
||||
case 0b00010001: {
|
||||
pmullt(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), Src2.Z());
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: {
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D());
|
||||
break;
|
||||
}
|
||||
case 0b00000001: {
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
|
||||
break;
|
||||
}
|
||||
case 0b00010000: {
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
|
||||
break;
|
||||
}
|
||||
case 0b00010001: {
|
||||
pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q());
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
|
||||
case 0b00000001:
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
|
||||
break;
|
||||
case 0b00010000:
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
|
||||
break;
|
||||
case 0b00010001: pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q()); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector); break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
glossary: Splatter ~ a code generator backend that concatenates configurable macros instead of doing isel
|
||||
glossary: Splatter ~ a code generator backend that concaternates configurable macros instead of doing isel
|
||||
glossary: IR ~ Intermediate Representation, our high-level opcode representation, loosely modeling arm64
|
||||
glossary: SSA ~ Single Static Assignment, a form of representing IR in memory
|
||||
glossary: Basic Block ~ A block of instructions with no control flow, terminated by control flow
|
||||
@@ -68,10 +68,6 @@ PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
static void PrintMsg(const char* Value) {
|
||||
LogMan::Msg::DFmt("{}", Value);
|
||||
}
|
||||
|
||||
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
|
||||
}
|
||||
@@ -137,8 +133,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.S(), Src1.S());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -155,8 +151,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -180,8 +176,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
mov(ARMEmitter::Size::i32Bit, TMP2, Src1);
|
||||
}
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -198,8 +194,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -216,8 +212,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -234,8 +230,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -258,8 +254,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -280,8 +276,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -298,8 +294,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -316,8 +312,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -334,8 +330,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -355,8 +351,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -373,8 +369,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -398,8 +394,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -420,8 +416,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -438,8 +434,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
// tmp2 (x1/x11): source 2
|
||||
// tmp3 (x2/x12): source 3
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
|
||||
stp<ARMEmitter::IndexType::PRE>(TMP1, ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
@@ -480,8 +476,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
movz(ARMEmitter::Size::i32Bit, TMP1, Control);
|
||||
|
||||
ldr(TMP2, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP2);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -497,7 +493,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
}
|
||||
}
|
||||
|
||||
static void DirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record, bool Call) {
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record, bool Call) {
|
||||
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
|
||||
uintptr_t CallerAddress = JumpThunkStartAddress + Record->CallerOffset;
|
||||
auto BranchOffset = JumpThunkStartAddress / 4 - CallerAddress / 4;
|
||||
@@ -515,12 +511,11 @@ static void DirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record,
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(CallerAddress), 4);
|
||||
}
|
||||
|
||||
static void IndirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
|
||||
uint32_t BranchInst = 0;
|
||||
ARMEmitter::Emitter BranchEmit(reinterpret_cast<uint8_t*>(&BranchInst), 4);
|
||||
// Restore branch +2 instructions to jump to the linker block
|
||||
BranchEmit.b(0x2);
|
||||
BranchEmit.b(0x8);
|
||||
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(BranchInst, std::memory_order::relaxed);
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
|
||||
@@ -537,13 +532,13 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
if (TFSet) {
|
||||
// If TF is set, the cache must be skipped as different code needs to be generated.
|
||||
Frame->State.rip = GuestRip;
|
||||
return Frame->Pointers.DispatcherLoopTop;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
} else {
|
||||
{
|
||||
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
|
||||
auto lk_inval =
|
||||
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
HostCode = Thread->LookupCache->FindBlock(Thread, GuestRip);
|
||||
HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
}
|
||||
if (!HostCode) {
|
||||
// Hold a reference to the code buffer, to avoid linking unmapped code if compilation triggers a recreation.
|
||||
@@ -568,7 +563,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
// Lock here is necessary to prevent simultaneous linking and delinking
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
|
||||
// For non-calls, this would extend into the block's code, however that's fine as an out-of-range adr would never
|
||||
// be generated avoiding any false positives.
|
||||
@@ -581,12 +576,14 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
|
||||
if (KnownCallMarkerInst == ExpectedKnownCallMarkerInst) {
|
||||
BranchEmit.bl(BranchOffset);
|
||||
Thread->LookupCache->AddBlockLink(
|
||||
GuestRip, Record, [](FEXCore::Context::ExitFunctionLinkData* Record) { DirectBlockDelinker(Record, true); }, lk);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
DirectBlockDelinker(Frame, Record, true);
|
||||
});
|
||||
} else {
|
||||
BranchEmit.b(BranchOffset);
|
||||
Thread->LookupCache->AddBlockLink(
|
||||
GuestRip, Record, [](FEXCore::Context::ExitFunctionLinkData* Record) { DirectBlockDelinker(Record, false); }, lk);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
DirectBlockDelinker(Frame, Record, false);
|
||||
});
|
||||
}
|
||||
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(CallerAddress)).store(BranchInst, std::memory_order::relaxed);
|
||||
@@ -594,7 +591,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
} else {
|
||||
// This case is common between calls and jumps as the thunk callsite can be left untouched.
|
||||
std::atomic_ref<uint64_t>(Record->HostCode).store(HostCode, std::memory_order::seq_cst);
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
#ifdef _M_ARM_64
|
||||
// Make memory write visible to other threads reading the same location
|
||||
asm volatile("dc cvau, %0; dsb ish" : : "r"(Record->HostCode) :);
|
||||
#endif
|
||||
@@ -605,7 +602,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(LdrInst, std::memory_order::relaxed);
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker, lk);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker);
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
@@ -616,67 +613,78 @@ void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::Ref Node) {}
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: CPUBackend(*ctx, Thread)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128 != 0}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256 != 0}
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsAVX256 {ctx->HostFeatures.SupportsAVX && ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES != 0}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP != 0}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP}
|
||||
, CTX {ctx}
|
||||
, TempCodeBufferAllocator(ctx->CPUBackendAllocator, 0) {
|
||||
, TempAllocator(ctx->CPUBackendAllocator, 0) {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
RAPass->AddRegisters(IR::RegClass::GPR, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::GPRFixed, StaticRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPR, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPRFixed, StaticFPRegisters.size());
|
||||
RAPass->SetNumPairRegs(PairRegisters);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->PairRegs = PairRegisters;
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
|
||||
// Common
|
||||
auto& Ptrs = ThreadState->CurrentFrame->Pointers;
|
||||
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
Ptrs.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Ptrs.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Ptrs.PrintMsgValue = reinterpret_cast<uint64_t>(PrintMsg);
|
||||
|
||||
Ptrs.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
|
||||
Ptrs.MonoBackpatcherWrite = reinterpret_cast<uint64_t>(&Context::ContextImpl::MonoBackpatcherWrite);
|
||||
Ptrs.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
|
||||
Common.MonoBackpatcherWrite = reinterpret_cast<uint64_t>(&Context::ContextImpl::MonoBackpatcherWrite);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Ptrs.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
|
||||
Ptrs.XCRFunction = PMF.GetConvertedPointer();
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::HLE::SyscallHandler::HandleSyscall);
|
||||
Ptrs.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Ptrs.SyscallHandlerFunc = PMF.GetVTableEntry(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = PMF.GetVTableEntry(CTX->SyscallHandler);
|
||||
}
|
||||
Ptrs.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore::ExitFunctionLink);
|
||||
Ptrs.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
Ptrs.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore::ExitFunctionLink);
|
||||
|
||||
// Platform Specific
|
||||
auto& AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
}
|
||||
|
||||
CurrentCodeBuffer = SharedCodeBuffers.GetLatest();
|
||||
CurrentCodeBuffer = CodeBuffers.GetLatest();
|
||||
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
const char JITString[] = "FEXJIT::Arm64JITCore::";
|
||||
EmitString(JITString);
|
||||
Align();
|
||||
}
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
auto lk = PrevCodeBuffer->LookupCache->AcquireWriteLock();
|
||||
std::lock_guard lk(PrevCodeBuffer->LookupCache->WriteLock);
|
||||
|
||||
auto CodeBuffer = AcquireNewSharedCodeBuffer();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CodeBuffer->LookupCache, lk);
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache);
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
@@ -753,7 +761,7 @@ void Arm64JITCore::EmitTFCheck() {
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGTRAP));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
|
||||
(void)Bind(&l_TFBlocked);
|
||||
@@ -767,23 +775,16 @@ void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
constexpr size_t InterruptPageOffset =
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState);
|
||||
if constexpr (InterruptPageOffset <= 32760) {
|
||||
str(ARMEmitter::XReg::zr, STATE, InterruptPageOffset);
|
||||
} else {
|
||||
// Need to use vector 128-bit store for this range.
|
||||
// Doesn't matter which register we use to store.
|
||||
str(ARMEmitter::QReg::q0, STATE, InterruptPageOffset);
|
||||
}
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::ForwardLabel l_NoSuspend;
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
(void)Bind(&l_NoSuspend);
|
||||
#endif
|
||||
@@ -809,62 +810,22 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
|
||||
|
||||
CodeBuffer::CodeBufferAllocation Arm64JITCore::AllocateCodeBufferInSharedCache(size_t Size) {
|
||||
CodeBuffer::CodeBufferAllocation AllocatedInfo {};
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
|
||||
"doesn't match up!\n");
|
||||
// Bring CodeBuffer up to date
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
auto lk = ThreadState->LookupCache->AcquireWriteLock();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
// Attempt to allocate a buffer from the SharedCodeBuffers.
|
||||
while (AllocatedInfo.BufferAllocationOffset == nullptr) {
|
||||
AllocatedInfo = CurrentCodeBuffer->AtomicAllocateBuffer(Size);
|
||||
|
||||
if (AllocatedInfo.BufferAllocationOffset == nullptr) {
|
||||
// If it didn't fit then clear the buffer and try again.
|
||||
// This has the possibility of migrating the SharedCodeBuffer. See above in `Arm64JITCore::ClearCache()`
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
return AllocatedInfo;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
const auto PrevNumAllocations = Relocations.size();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->DebugData = DebugData;
|
||||
this->IR = IR;
|
||||
RequiresFarARM64Jumps = false;
|
||||
SSANodeMultiplier = 24;
|
||||
|
||||
// Prepare restart via long jump in case branch encoding fails.
|
||||
// This uses UncheckedLongJump since we don't implement std::longjmp in WoA setups
|
||||
switch (static_cast<RestartOptions::Control>(FEXCore::UncheckedLongJump::SetJump(ThreadState->RestartJump))) {
|
||||
switch (static_cast<RestartOptions::Control>(FEXCore::LongJump::SetJump(RestartControl.RestartJump))) {
|
||||
case RestartOptions::Control::Incoming:
|
||||
// Nothing
|
||||
break;
|
||||
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
|
||||
case RestartOptions::Control::NeedsLargerJITSpace:
|
||||
// Get rid of the claimed buffer immediately, we can't fit in it at all.
|
||||
TempCodeBufferAllocator.UnclaimBuffer();
|
||||
SSANodeMultiplier *= 2;
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Arm64 restart condition!");
|
||||
default: ERROR_AND_DIE_FMT("Unhandled Arm64 restart condition!");
|
||||
}
|
||||
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
@@ -872,27 +833,18 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
CallReturnTargets.clear();
|
||||
PendingJumpThunks.clear();
|
||||
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
|
||||
Relocations.resize(PrevNumAllocations, FEXCore::CPU::Relocation::Default()); // Discard any relocations generated from a previous attempt
|
||||
|
||||
CodeData.EntryPoints.clear();
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
// One page baseline, plus SSANodeMultipler bytes, plus another page for guard page.
|
||||
const uint32_t DesiredBufferRange = AlignUp(FEXCore::Utils::FEX_PAGE_SIZE * 2 + SSACount * SSANodeMultiplier, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
uint32_t BufferRange = 0x1000 + SSACount * 24;
|
||||
|
||||
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
|
||||
// This minimizes lock contention of CodeBufferWriteMutex.
|
||||
auto TempCodeBufferInfo = TempCodeBufferAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
|
||||
auto TempCodeBuffer = TempCodeBufferInfo.Ptr;
|
||||
const uint32_t UsableBufferRange = TempCodeBufferInfo.Size - FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
|
||||
SetBuffer(TempCodeBuffer, UsableBufferRange);
|
||||
|
||||
ThreadState->JITGuardPage = reinterpret_cast<uintptr_t>(TempCodeBuffer) + UsableBufferRange;
|
||||
ThreadState->JITGuardOverflowArgument = FEXCore::ToUnderlying(RestartOptions::Control::NeedsLargerJITSpace);
|
||||
auto TempCodeBuffer = TempAllocator.ReownOrClaimBuffer(BufferRange);
|
||||
SetBuffer(TempCodeBuffer, BufferRange);
|
||||
|
||||
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
|
||||
LOGMAN_THROW_A_FMT(GetCursorOffset() == 0, "Needs to be zero");
|
||||
|
||||
// Put the code header at the start of the data block.
|
||||
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
|
||||
@@ -928,6 +880,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
PendingCallReturnTargetLabel = nullptr;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
@@ -979,6 +932,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, CodeNode); break
|
||||
#define REGISTER_OP(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, CodeNode); break
|
||||
|
||||
@@ -1019,34 +974,22 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// This is a ExitFunctionLinkData struct
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(0); // HostCode
|
||||
if (PendingJumpThunk.PatchSiteAddress) {
|
||||
// GuestRIP with an extra step
|
||||
PlaceNamedSymbolLiteral(
|
||||
InsertGuestPatchableRIPLiteral(PendingJumpThunk.GuestRIP, PendingJumpThunk.PatchSiteAddress, PendingJumpThunk.PatchSiteSize));
|
||||
} else {
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
|
||||
}
|
||||
dc64(0); // HostCode
|
||||
dc64(PendingJumpThunk.GuestRIP); // GuestRIP
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
}
|
||||
|
||||
BindOrRestart(&l_ExitLink);
|
||||
PlaceNamedSymbolLiteral(InsertNamedSymbolLiteral(RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER));
|
||||
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
|
||||
// CodeSize not including the header or tail data.
|
||||
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeBegin;
|
||||
|
||||
// Add the JitCodeTail (written later)
|
||||
// Add the JitCodeTail
|
||||
Align(alignof(JITCodeTail));
|
||||
const auto JITBlockTailLocation = GetCursorAddress<uint8_t*>();
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
JITCodeTail JITBlockTail {
|
||||
.RIP = Entry,
|
||||
.GuestSize = Size,
|
||||
.SpinLockFutex = 0,
|
||||
.SingleInst = SingleInst,
|
||||
};
|
||||
auto JITBlockTailLocation = GetCursorAddress<uint8_t*>();
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
@@ -1064,13 +1007,23 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
// FEXCore::Utils::vl64 GuestRIPOffset;
|
||||
// };
|
||||
|
||||
const auto JITRIPEntriesBegin = JITBlockTailLocation + sizeof(JITBlockTail);
|
||||
auto JITRIPEntriesBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
JITBlockTail->GuestSize = Size;
|
||||
JITBlockTail->SingleInst = SingleInst;
|
||||
JITBlockTail->SpinLockFutex = 0;
|
||||
|
||||
auto JITRIPEntriesLocation = JITRIPEntriesBegin;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail.NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail.OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
|
||||
@@ -1086,48 +1039,61 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
}
|
||||
|
||||
SetCursorOffset(JITRIPEntriesLocation - CodeData.BlockBegin);
|
||||
// Make sure code is 16B aligned on the tail.
|
||||
// Can't use Align16B here as vl64pair can cause non-4byte alignment.
|
||||
Align(16);
|
||||
CursorIncrement(JITRIPEntriesLocation - JITRIPEntriesBegin);
|
||||
Align();
|
||||
|
||||
// Beginning of emission is guaranteed to be offset zero. So the code data size is just the current cursor offset.
|
||||
CodeData.Size = GetCursorOffset();
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
// Finalize and write block tail data
|
||||
JITBlockTail.Size = CodeData.Size;
|
||||
{
|
||||
memcpy(JITBlockTailLocation, &JITBlockTail, sizeof(JITBlockTail));
|
||||
SetCursorOffset(JITBlockTailLocation - CodeData.BlockBegin + offsetof(JITCodeTail, RIP));
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(JITBlockTail.RIP));
|
||||
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
// Emitter buffer is no longer used, guard against misuse by setting to nullptr.
|
||||
SetBuffer(nullptr, 0);
|
||||
}
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
// Migrate the compile output from temporary storage to the actual CodeBuffer.
|
||||
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
|
||||
{
|
||||
LOGMAN_THROW_A_FMT(CodeData.Size % 16 == 0, "Needs to be 16B aligned!");
|
||||
auto CodeBufferLock = std::unique_lock {CodeBuffers.CodeBufferWriteMutex};
|
||||
|
||||
auto AllocatedInfo = AllocateCodeBufferInSharedCache(CodeData.Size);
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
LOGMAN_THROW_A_FMT((reinterpret_cast<uintptr_t>(AllocatedInfo.BufferAllocationOffset) % 16) == 0, "Allocated buffer wasn't 16B "
|
||||
"aligned?");
|
||||
// Query size of generated code
|
||||
const auto TempSize = GetCursorOffset();
|
||||
LOGMAN_THROW_A_FMT(TempSize <= BufferRange, "Exceeded bounds of temporary buffer ({:#x} vs {:#x})", TempSize, BufferRange);
|
||||
|
||||
// Bring CodeBuffer up to date
|
||||
{
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
|
||||
"doesn't match up!\n");
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
|
||||
}
|
||||
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
SetBuffer(CurrentCodeBuffer->Ptr, CurrentCodeBuffer->Size);
|
||||
SetCursorOffset(AlignUp(CodeBuffers.LatestOffset, 16));
|
||||
if ((GetCursorOffset() + TempSize) > (CurrentCodeBuffer->Size - Utils::FEX_PAGE_SIZE)) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
Align16B();
|
||||
|
||||
CodeBuffers.LatestOffset = GetCursorOffset();
|
||||
}
|
||||
|
||||
// Adjust host addresses
|
||||
const auto Delta = AllocatedInfo.BufferAllocationOffset - CodeData.BlockBegin;
|
||||
const auto Delta = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
CodeData.BlockBegin += Delta;
|
||||
for (auto& EntryPoint : CodeData.EntryPoints) {
|
||||
EntryPoint.second += Delta;
|
||||
}
|
||||
CodeBegin += Delta;
|
||||
CodeData.HostCodeOffset = CodeData.BlockBegin - CurrentCodeBuffer->GetBufferBase();
|
||||
|
||||
// Copy over CodeBuffer contents
|
||||
memcpy(AllocatedInfo.BufferAllocationOffset, TempCodeBuffer, CodeData.Size);
|
||||
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
|
||||
SetCursorOffset(CodeBuffers.LatestOffset + TempSize);
|
||||
|
||||
CodeBuffers.LatestOffset = GetCursorOffset();
|
||||
}
|
||||
|
||||
TempCodeBufferAllocator.DelayedDisownBuffer();
|
||||
TempAllocator.DelayedDisownBuffer();
|
||||
|
||||
ClearICache(CodeBegin, CodeOnlySize);
|
||||
|
||||
@@ -1163,22 +1129,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
return std::move(CodeData);
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::LoadCachedCode(std::span<const uint8_t> HostBytes) {
|
||||
// we stored it aligned, better still be?
|
||||
LOGMAN_THROW_A_FMT(HostBytes.size() % 16 == 0, "Needs to be 16B aligned!");
|
||||
auto AllocatedInfo = AllocateCodeBufferInSharedCache(HostBytes.size());
|
||||
|
||||
uint8_t* Dest = AllocatedInfo.BufferAllocationOffset;
|
||||
memcpy(Dest, HostBytes.data(), HostBytes.size());
|
||||
ClearICache(Dest, HostBytes.size());
|
||||
|
||||
CPUBackend::CompiledCode Result;
|
||||
Result.BlockBegin = Dest;
|
||||
Result.Size = HostBytes.size();
|
||||
Result.HostCodeOffset = Dest - CurrentCodeBuffer->GetBufferBase();
|
||||
return Result;
|
||||
}
|
||||
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0) {
|
||||
return;
|
||||
|
||||
@@ -10,17 +10,13 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
@@ -30,19 +26,16 @@ $end_info$
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct ExitFunctionLinkData;
|
||||
}
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationPass;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
@@ -54,9 +47,6 @@ public:
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) override;
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode LoadCachedCode(std::span<const uint8_t> HostBytes) override;
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
void ClearRelocations() override {
|
||||
@@ -64,6 +54,9 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
|
||||
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
const bool HostSupportsAVX256 {};
|
||||
@@ -71,10 +64,10 @@ private:
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
struct RestartOptions {
|
||||
FEXCore::LongJump::JumpBuf RestartJump;
|
||||
enum class Control : uint64_t {
|
||||
Incoming = 0,
|
||||
EnableFarARM64Jumps = 1,
|
||||
NeedsLargerJITSpace = 2,
|
||||
};
|
||||
};
|
||||
|
||||
@@ -82,8 +75,6 @@ private:
|
||||
// In the rare case when those assumptions are broken, FEX needs to safely restart the JIT.
|
||||
RestartOptions RestartControl {};
|
||||
bool RequiresFarARM64Jumps {};
|
||||
// Default to 6 instructions per SSA node.
|
||||
uint32_t SSANodeMultiplier {24};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
|
||||
ARMEmitter::BiDirectionalLabel* PendingCallReturnTargetLabel {};
|
||||
@@ -105,24 +96,20 @@ private:
|
||||
uint64_t CallerAddress;
|
||||
uint64_t GuestRIP;
|
||||
ARMEmitter::ForwardLabel Label;
|
||||
uint64_t PatchSiteAddress = 0;
|
||||
uint8_t PatchSiteSize = 0;
|
||||
};
|
||||
fextl::vector<PendingJumpThunk> PendingJumpThunks;
|
||||
|
||||
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempCodeBufferAllocator;
|
||||
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempAllocator;
|
||||
|
||||
static uint64_t ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const {
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::GPRFixed || RegClass == IR::RegClass::GPR, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (RegClass == IR::RegClass::GPRFixed) {
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return StaticRegisters[Reg.Reg];
|
||||
} else if (RegClass == IR::RegClass::GPR) {
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return GeneralRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
@@ -141,13 +128,11 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VRegister GetVReg(IR::PhysicalRegister Reg) const {
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::FPRFixed || RegClass == IR::RegClass::FPR, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (RegClass == IR::RegClass::FPRFixed) {
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
} else if (RegClass == IR::RegClass::FPR) {
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return GeneralFPRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
@@ -165,8 +150,8 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static IR::RegClass GetRegClass(IR::Ref Node) {
|
||||
return IR::PhysicalRegister(Node).AsRegClass();
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::Ref Node) const {
|
||||
return FEXCore::IR::RegisterClassType {IR::PhysicalRegister(Node).Class};
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -183,7 +168,7 @@ private:
|
||||
// Converts IR-base shift type to ARMEmitter shift type.
|
||||
// Will be a no-op, only a type conversion since the two definitions match.
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) {
|
||||
ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
|
||||
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
|
||||
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
|
||||
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
|
||||
@@ -191,23 +176,18 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
return Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSize(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::Size ConvertSize(IR::OpSize Size) {
|
||||
return Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
@@ -219,105 +199,105 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize16(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSize16(ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize8(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
|
||||
return ARMEmitter::ToVectorSizePair(ConvertSubRegSize16(Op));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair16(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::Condition MapCC(IR::CondClass Cond) {
|
||||
switch (Cond) {
|
||||
case IR::CondClass::EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case IR::CondClass::NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case IR::CondClass::SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case IR::CondClass::SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case IR::CondClass::SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case IR::CondClass::SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case IR::CondClass::UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case IR::CondClass::ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case IR::CondClass::UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case IR::CondClass::ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case IR::CondClass::FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case IR::CondClass::FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case IR::CondClass::FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case IR::CondClass::FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case IR::CondClass::FU:
|
||||
case IR::CondClass::VS: return ARMEmitter::Condition::CC_VS;
|
||||
case IR::CondClass::FNU:
|
||||
case IR::CondClass::VC: return ARMEmitter::Condition::CC_VC;
|
||||
case IR::CondClass::MI: return ARMEmitter::Condition::CC_MI;
|
||||
case IR::CondClass::PL: return ARMEmitter::Condition::CC_PL;
|
||||
ARMEmitter::Condition MapCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU:
|
||||
case FEXCore::IR::COND_VS: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU:
|
||||
case FEXCore::IR::COND_VC: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsFPR(IR::RegClass Class) {
|
||||
return Class == IR::RegClass::FPR || Class == IR::RegClass::FPRFixed;
|
||||
bool IsFPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsGPR(IR::RegClass Class) {
|
||||
return Class == IR::RegClass::GPR || Class == IR::RegClass::GPRFixed;
|
||||
bool IsGPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsGPR(IR::Ref Node) {
|
||||
bool IsGPR(IR::Ref Node) {
|
||||
return IsGPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsFPR(IR::Ref Node) {
|
||||
bool IsFPR(IR::Ref Node) {
|
||||
return IsFPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::PhysicalRegister(Wrap).AsRegClass());
|
||||
bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::PhysicalRegister(Wrap).AsRegClass());
|
||||
bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -347,8 +327,8 @@ private:
|
||||
uint32_t End;
|
||||
};
|
||||
|
||||
void EmitLinkedBranch(uint64_t GuestRIP, bool Call, uint64_t PatchSiteAddress = 0, uint8_t PatchSiteSize = 0) {
|
||||
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}, PatchSiteAddress, PatchSiteSize});
|
||||
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
|
||||
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
|
||||
auto& Thunk = PendingJumpThunks.back();
|
||||
BindOrRestart(&Thunk.Label);
|
||||
if (Call) {
|
||||
@@ -359,7 +339,9 @@ private:
|
||||
}
|
||||
|
||||
// Restart helpers
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void bl_OrRestart(T* Label) {
|
||||
if (bl(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
@@ -367,10 +349,12 @@ private:
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void b_OrRestart(T* Label) {
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
@@ -378,10 +362,12 @@ private:
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void b_OrRestart(ARMEmitter::Condition Cond, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -399,10 +385,12 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void cbz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -420,10 +408,12 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void cbnz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -441,10 +431,12 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void tbz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -462,10 +454,12 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void tbnz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -483,40 +477,38 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void adr_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
if (LongAddressGen(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Unable to encode long ADR.");
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (adr(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Long ADR currently unsupported!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void adrp_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
if (LongAddressGen(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Unable to encode long ADRP.");
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (adrp(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Long ADRP currently unsupported!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void BindOrRestart(T* Label) {
|
||||
if (Bind(Label)) {
|
||||
return;
|
||||
@@ -524,13 +516,15 @@ private:
|
||||
|
||||
if (RequiresFarARM64Jumps) {
|
||||
// This should have been caught before this point.
|
||||
ERROR_AND_DIE_FMT("Unhandled long bind");
|
||||
ERROR_AND_DIE_FMT("Oops. Unhandled long bind.");
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
|
||||
@@ -539,6 +533,8 @@ private:
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
@@ -564,9 +560,6 @@ private:
|
||||
*/
|
||||
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
|
||||
|
||||
void InsertGuestPatchableDataMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize);
|
||||
void InsertGuestPatchableRIPMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
@@ -578,35 +571,19 @@ private:
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Inserts a relocation for a constant value relative to the guest entrypoint
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertGuestRIPLiteral(uint64_t GuestRIP);
|
||||
|
||||
/**
|
||||
* @brief Like InsertGuestRIPLiteral, but with patch information to recompute value from live guest bytes at cache load time
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit);
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
/**
|
||||
* Returns any relocations generated since the last call to TakeRelocations.
|
||||
*
|
||||
* GuestBaseAddress must match the base virtual address to which the
|
||||
* input x86 binary is mapped.
|
||||
*/
|
||||
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) override;
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation>);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() override;
|
||||
|
||||
/** @} */
|
||||
|
||||
@@ -629,7 +606,7 @@ private:
|
||||
void Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
|
||||
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize);
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
|
||||
void EmitTFCheck();
|
||||
|
||||
@@ -637,8 +614,6 @@ private:
|
||||
|
||||
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
|
||||
|
||||
[[nodiscard]] CodeBuffer::CodeBufferAllocation AllocateCodeBufferInSharedCache(size_t Size);
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
///< Unhandled handler
|
||||
|
||||
@@ -21,7 +21,7 @@ DEF_OP(LoadContext) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -52,7 +52,7 @@ DEF_OP(LoadContext) {
|
||||
DEF_OP(LoadContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
@@ -78,7 +78,7 @@ DEF_OP(StoreContext) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -110,7 +110,7 @@ DEF_OP(StoreContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto Src1 = GetZeroableReg(Op->Value1);
|
||||
auto Src2 = GetZeroableReg(Op->Value2);
|
||||
|
||||
@@ -135,11 +135,11 @@ DEF_OP(StoreContextPair) {
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
|
||||
|
||||
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
|
||||
} else if (Op->Class == IR::RegClass::FPR) {
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
@@ -175,13 +175,12 @@ DEF_OP(LoadAF) {
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto Reg = IR::PhysicalRegister(Node);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
auto Reg = IR::PhysicalRegister(Node);
|
||||
|
||||
if (RegClass == IR::RegClass::GPRFixed) {
|
||||
if (Reg.Class == IR::GPRFixedClass) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Reg), GetReg(Op->Value));
|
||||
} else if (RegClass == IR::RegClass::FPRFixed) {
|
||||
} else if (Reg.Class == IR::FPRFixedClass) {
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
@@ -194,7 +193,7 @@ DEF_OP(StoreRegister) {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", RegClass);
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Reg.Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -226,7 +225,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
@@ -267,7 +266,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
ldr(Dst.Q(), TMP1, Op->BaseOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, Op->BaseOffset);
|
||||
ldur(Dst.Q(), TMP1);
|
||||
ldur(Dst.Q(), TMP1, Op->BaseOffset);
|
||||
}
|
||||
break;
|
||||
case IR::OpSize::i256Bit:
|
||||
@@ -289,7 +288,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Value = GetReg(Op->Value);
|
||||
|
||||
switch (Op->Stride) {
|
||||
@@ -333,7 +332,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
str(Value.Q(), TMP1, Op->BaseOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, Op->BaseOffset);
|
||||
stur(Value.Q(), TMP1);
|
||||
stur(Value.Q(), TMP1, Op->BaseOffset);
|
||||
}
|
||||
break;
|
||||
case IR::OpSize::i256Bit:
|
||||
@@ -373,7 +372,7 @@ DEF_OP(SpillRegister) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -414,7 +413,7 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -453,7 +452,7 @@ DEF_OP(SpillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class);
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -462,7 +461,7 @@ DEF_OP(FillRegister) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -503,7 +502,7 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -542,7 +541,7 @@ DEF_OP(FillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class);
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -563,25 +562,7 @@ DEF_OP(LoadDF) {
|
||||
auto Flag = X86State::RFLAG_DF_RAW_LOC;
|
||||
|
||||
// DF needs sign extension to turn 0x1/0xFF into 1/-1
|
||||
ldrsb(Dst.X(), STATE, ARRAY_OFFSETOF(FEXCore::Core::CPUState, flags, Flag));
|
||||
}
|
||||
|
||||
DEF_OP(ContextClear) {
|
||||
auto Op = IROp->C<IR::IROp_ContextClear>();
|
||||
if (CTX->HostFeatures.PreferZVAForVZero) {
|
||||
// We can use CLZero directly when hardware supports it.
|
||||
// Provides a fairly generous speed-up on Ampere1A hardware.
|
||||
// TODO: When FEAT_MOPS hardware ships, test memset using MOPS.
|
||||
for (size_t i = 0; i < Op->Size; i += 64) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE.R(), Op->Offset + i);
|
||||
dc(ARMEmitter::DataCacheOperation::ZVA, TMP1);
|
||||
}
|
||||
} else {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
for (size_t i = 0; i < Op->Size; i += 32) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(VTMP1.Q(), VTMP1.Q(), STATE.R(), Op->Offset + i);
|
||||
}
|
||||
}
|
||||
ldrsb(Dst.X(), STATE, offsetof(FEXCore::Core::CPUState, flags[Flag]));
|
||||
}
|
||||
|
||||
ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
@@ -597,14 +578,14 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, Const);
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType) {
|
||||
case IR::MemOffsetType::SXTX:
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
case IR::MemOffsetType::UXTW:
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
case IR::MemOffsetType::SXTW:
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -631,20 +612,20 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType) {
|
||||
case IR::MemOffsetType::SXTX:
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MemOffsetType::UXTW:
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MemOffsetType::SXTW:
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType.Val); break;
|
||||
}
|
||||
}
|
||||
return Tmp;
|
||||
@@ -695,7 +676,7 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
|
||||
// Note that we do nothing with the offset type and offset scale,
|
||||
// since SVE loads and stores don't have the ability to perform an
|
||||
// optional extension or shift as part of their behavior.
|
||||
LOGMAN_THROW_A_FMT(OffsetType == IR::MemOffsetType::SXTX, "Currently only the default offset type (SXTX) is supported.");
|
||||
LOGMAN_THROW_A_FMT(OffsetType.Val == IR::MEM_OFFSET_SXTX.Val, "Currently only the default offset type (SXTX) is supported.");
|
||||
|
||||
const auto RegOffset = GetReg(Offset);
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
|
||||
@@ -708,7 +689,7 @@ DEF_OP(LoadMem) {
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -742,7 +723,7 @@ DEF_OP(LoadMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
@@ -770,13 +751,13 @@ DEF_OP(LoadMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
@@ -795,10 +776,12 @@ DEF_OP(LoadMemTSO) {
|
||||
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == IR::RegClass::GPR) {
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -810,10 +793,12 @@ DEF_OP(LoadMemTSO) {
|
||||
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -825,8 +810,10 @@ DEF_OP(LoadMemTSO) {
|
||||
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1058,7 +1045,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
ARMEmitter::VRegister IncomingDst, std::optional<ARMEmitter::Register> BaseAddr,
|
||||
ARMEmitter::VRegister VectorIndexLow, std::optional<ARMEmitter::VRegister> VectorIndexHigh,
|
||||
ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize, size_t DataElementOffsetStart,
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize) {
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid element size");
|
||||
|
||||
const auto PerformSMove = [this](IR::OpSize ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
@@ -1134,17 +1121,17 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
// Calculate memory position for this gather load
|
||||
if (BaseAddr.has_value()) {
|
||||
if (VectorIndexSize == IR::OpSize::i32Bit) {
|
||||
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
} else {
|
||||
///< In this case we have no base address, All addresses come from the vector register itself
|
||||
if (VectorIndexSize == IR::OpSize::i32Bit) {
|
||||
// Sign extend and shift in to the 64-bit register
|
||||
sbfiz(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
|
||||
sbfiz(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
|
||||
} else {
|
||||
lsl(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
|
||||
lsl(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1202,8 +1189,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
const bool SupportsSVELoad = (HostSupportsSVE128 || HostSupportsSVE256) &&
|
||||
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) &&
|
||||
VectorIndexSize == IROp->ElementSize && Op->AddrSize == IR::OpSize::i64Bit;
|
||||
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) && VectorIndexSize == IROp->ElementSize;
|
||||
|
||||
if (SupportsSVELoad) {
|
||||
uint8_t SVEScale = FEXCore::ilog2(OffsetScale);
|
||||
@@ -1261,7 +1247,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit, "Can't emulate this gather load in the backend! Programming error!");
|
||||
Emulate128BitGather(IROp->Size, IROp->ElementSize, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
|
||||
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale, Op->AddrSize);
|
||||
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1286,9 +1272,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
const bool SupportsSVELoad = HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4) && Op->AddrSize == IR::OpSize::i64Bit;
|
||||
|
||||
if (SupportsSVELoad) {
|
||||
if (HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4)) {
|
||||
ARMEmitter::SVEModType ModType = ARMEmitter::SVEModType::MOD_NONE;
|
||||
if (OffsetScale != 1) {
|
||||
ModType = ARMEmitter::SVEModType::MOD_LSL;
|
||||
@@ -1342,7 +1326,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
}
|
||||
} else {
|
||||
Emulate128BitGather(IR::OpSize::i128Bit, IR::OpSize::i32Bit, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
|
||||
IR::OpSize::i64Bit, 0, 0, OffsetScale, Op->AddrSize);
|
||||
IR::OpSize::i64Bit, 0, 0, OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1643,7 +1627,7 @@ DEF_OP(StoreMem) {
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
|
||||
@@ -1754,7 +1738,7 @@ DEF_OP(StoreMemPair) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src1 = GetZeroableReg(Op->Value1);
|
||||
const auto Src2 = GetZeroableReg(Op->Value2);
|
||||
switch (OpSize) {
|
||||
@@ -1781,13 +1765,13 @@ DEF_OP(StoreMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
@@ -1799,8 +1783,10 @@ DEF_OP(StoreMemTSO) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
} else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
|
||||
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
|
||||
@@ -1808,15 +1794,17 @@ DEF_OP(StoreMemTSO) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(Src, MemReg);
|
||||
} else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
|
||||
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
|
||||
@@ -1849,6 +1837,13 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
|
||||
DEF_OP(MemSet) {
|
||||
// TODO: A future looking task would be to support this with ARM's MOPS instructions.
|
||||
// The 8-bit non-atomic forward path directly matches ARM's SETP/SETM/SETE instruction,
|
||||
// while the backward version needs some fixup to convert it to a forward direction.
|
||||
//
|
||||
// Assuming non-atomicity and non-faulting behaviour, this can accelerate this implementation.
|
||||
// Additionally: This is commonly used as a memset to zero. If we know up-front with an inline constant
|
||||
// that the value is zero, we can optimize any operation larger than 8-bit down to 8-bit to use the MOPS implementation.
|
||||
const auto Op = IROp->C<IR::IROp_MemSet>();
|
||||
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
@@ -1903,7 +1898,9 @@ DEF_OP(MemSet) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(Value.W(), TMP2);
|
||||
} else {
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
nop();
|
||||
}
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(Value.W(), TMP2); break;
|
||||
case 4: stlr(Value.W(), TMP2); break;
|
||||
@@ -1926,30 +1923,8 @@ DEF_OP(MemSet) {
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
auto EmitMemset = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
const bool IsBackwards = Direction == -1;
|
||||
|
||||
// Sets the result to the final address written depending on
|
||||
// whether or not the memset is forwards or backwards.
|
||||
const auto MakeFinalAddress = [&] {
|
||||
if (IsBackwards) {
|
||||
switch (Size) {
|
||||
case 1: sub(Dst.X(), MemReg.X(), Length.X()); break;
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: sub(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled MemSet size: {}", Size); break;
|
||||
}
|
||||
} else {
|
||||
switch (Size) {
|
||||
case 1: add(Dst.X(), MemReg.X(), Length.X()); break;
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: add(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled MemSet size: {}", Size); break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel AgainInternal {};
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
@@ -1958,56 +1933,12 @@ DEF_OP(MemSet) {
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
if (CTX->HostFeatures.SupportsMOPS) {
|
||||
const bool Is8Bit = SubRegSize == ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
// We can handle 8-bit memsets and any other size that happens
|
||||
// to be using an inlined zero value (resulting in the use of ZR).
|
||||
//
|
||||
// NOTE:
|
||||
// Strictly speaking, this can also be trivially expanded to handle other sizes
|
||||
// that happen to use any value that could fit inside a byte if the need
|
||||
// arises. This does increase branching and code generation, however, since
|
||||
// we'd still need to emit the fallback in the event a value for a larger size
|
||||
// falls outside the range of a byte instead of only generating the MOPS code.
|
||||
if (Is8Bit || Value == ARMEmitter::Reg::zr) {
|
||||
// If we're performing a non-byte-sized zeroing operation then we need to
|
||||
// scale the counter accordingly. (e.g. a 64-bit memset of size 2 needs to
|
||||
// be turned into an 8-bit memset of size 16)
|
||||
if (!Is8Bit) {
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP1, TMP1, FEXCore::ToUnderlying(SubRegSize));
|
||||
}
|
||||
|
||||
// If backwards, then we need to adjust the starting address because
|
||||
// set{p, m, e} memset forwards, so we need to slide this bad boy
|
||||
// back like: (address - count) + 1.
|
||||
//
|
||||
// This lets us offset the address such that we can treat a backwards
|
||||
// memset as if it were a forwards one.
|
||||
if (IsBackwards) {
|
||||
sub(TMP2, TMP2, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
}
|
||||
|
||||
// Unfortunately set operations fiddle with NZCV, so we need to preserve it.
|
||||
mrs(TMP3, ARMEmitter::SystemRegister::NZCV);
|
||||
setp(TMP2, TMP1, Value.X());
|
||||
setm(TMP2, TMP1, Value.X());
|
||||
sete(TMP2, TMP1, Value.X());
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
|
||||
|
||||
MakeFinalAddress();
|
||||
(void)Bind(&DoneInternal);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
ARMEmitter::ForwardLabel AgainInternal256Exit {};
|
||||
ARMEmitter::BackwardLabel AgainInternal256 {};
|
||||
ARMEmitter::ForwardLabel AgainInternal128Exit {};
|
||||
ARMEmitter::BackwardLabel AgainInternal128 {};
|
||||
|
||||
if (IsBackwards) {
|
||||
if (Direction == -1) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
}
|
||||
|
||||
@@ -2045,23 +1976,39 @@ DEF_OP(MemSet) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (IsBackwards) {
|
||||
if (Direction == -1) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
}
|
||||
}
|
||||
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemStoreTSO(Value, Size, SizeDirection);
|
||||
MemStoreTSO(Value, OpSize, SizeDirection);
|
||||
} else {
|
||||
MemStore(Value, Size, SizeDirection);
|
||||
MemStore(Value, OpSize, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
|
||||
(void)Bind(&DoneInternal);
|
||||
|
||||
MakeFinalAddress();
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
case 1: add(Dst.X(), MemReg.X(), Length.X()); break;
|
||||
case 2: add(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 1); break;
|
||||
case 4: add(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 2); break;
|
||||
case 8: add(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 3); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case 1: sub(Dst.X(), MemReg.X(), Length.X()); break;
|
||||
case 2: sub(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 1); break;
|
||||
case 4: sub(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 2); break;
|
||||
case 8: sub(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 3); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (DirectionIsInline) {
|
||||
@@ -2084,6 +2031,10 @@ DEF_OP(MemSet) {
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
// TODO: A future looking task would be to support this with ARM's MOPS instructions.
|
||||
// The 8-bit non-atomic path directly matches ARM's CPYP/CPYM/CPYE instruction,
|
||||
//
|
||||
// Assuming non-atomicity and non-faulting behaviour, this can accelerate this implementation.
|
||||
const auto Op = IROp->C<IR::IROp_MemCpy>();
|
||||
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
@@ -2167,9 +2118,11 @@ DEF_OP(MemCpy) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
@@ -2191,9 +2144,11 @@ DEF_OP(MemCpy) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
@@ -2214,40 +2169,8 @@ DEF_OP(MemCpy) {
|
||||
};
|
||||
|
||||
auto EmitMemcpy = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
const bool IsBackwards = Direction == -1;
|
||||
|
||||
const auto FinalizeAddresses = [&] {
|
||||
if (IsBackwards) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
sub(Dst0.X(), TMP1, TMP3);
|
||||
sub(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
case 4:
|
||||
case 8:
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size));
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled MemCpy size: {}", Size); break;
|
||||
}
|
||||
} else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
add(Dst0.X(), TMP1, TMP3);
|
||||
add(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
case 4:
|
||||
case 8:
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size));
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled MemCpy size: {}", Size); break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel AgainInternal {};
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
@@ -2256,48 +2179,6 @@ DEF_OP(MemCpy) {
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
if (CTX->HostFeatures.SupportsMOPS) {
|
||||
// In the event we have an overlap (gross), we need to fall back
|
||||
// to the non-mops copy handler. Since the overlap check needs to
|
||||
// make use of NZCV, we need to save it. This can be avoided with
|
||||
// ARMv9.6+'s FEAT_CMPBR, but alas, we don't have access to that right now.
|
||||
//
|
||||
// NOTE: That we need to temporarily trash TMP1 and restore it after the
|
||||
// comparison.
|
||||
ARMEmitter::ForwardLabel OverlapCase;
|
||||
mrs(TMP4, ARMEmitter::SystemRegister::NZCV);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP2, TMP3);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP1, Length.X());
|
||||
mov(TMP1, Length.X());
|
||||
(void)bc(ARMEmitter::Condition::CC_LT, &OverlapCase);
|
||||
|
||||
// If doing something larger than a byte copy, then we need to scale
|
||||
// the counter value accordingly to convert it to bytes.
|
||||
if (Size > 1) {
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP1, TMP1, FEXCore::ilog2(Size));
|
||||
}
|
||||
|
||||
// Adjust addresses so that we treat the backward copy as a forward copy
|
||||
if (IsBackwards) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, TMP1);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP3, TMP3, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, Size);
|
||||
add(ARMEmitter::Size::i64Bit, TMP3, TMP3, Size);
|
||||
}
|
||||
|
||||
// Unfortunately copy operations fiddle with NZCV, so we need to preserve it.
|
||||
cpyfp(TMP2, TMP3, TMP1);
|
||||
cpyfm(TMP2, TMP3, TMP1);
|
||||
cpyfe(TMP2, TMP3, TMP1);
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP4);
|
||||
|
||||
(void)b(&DoneInternal);
|
||||
|
||||
// Turns out we overlap and need to fall back. Make sure to restore NZCV.
|
||||
(void)Bind(&OverlapCase);
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP4);
|
||||
}
|
||||
|
||||
ARMEmitter::ForwardLabel AbsPos {};
|
||||
ARMEmitter::ForwardLabel AgainInternal256Exit {};
|
||||
ARMEmitter::ForwardLabel AgainInternal128Exit {};
|
||||
@@ -2311,7 +2192,7 @@ DEF_OP(MemCpy) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
|
||||
(void)tbnz(TMP4, 63, &AgainInternal);
|
||||
|
||||
if (IsBackwards) {
|
||||
if (Direction == -1) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32 - Size);
|
||||
}
|
||||
@@ -2346,7 +2227,7 @@ DEF_OP(MemCpy) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (IsBackwards) {
|
||||
if (Direction == -1) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
add(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32 - Size);
|
||||
}
|
||||
@@ -2354,9 +2235,9 @@ DEF_OP(MemCpy) {
|
||||
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemCpyTSO(Size, SizeDirection);
|
||||
MemCpyTSO(OpSize, SizeDirection);
|
||||
} else {
|
||||
MemCpy(Size, SizeDirection);
|
||||
MemCpy(OpSize, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
@@ -2368,14 +2249,54 @@ DEF_OP(MemCpy) {
|
||||
mov(TMP2, MemRegSrc.X());
|
||||
mov(TMP3, Length.X());
|
||||
|
||||
FinalizeAddresses();
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
add(Dst0.X(), TMP1, TMP3);
|
||||
add(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
sub(Dst0.X(), TMP1, TMP3);
|
||||
sub(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (DirectionIsInline) {
|
||||
LOGMAN_THROW_A_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
EmitMemcpy(DirectionConstant);
|
||||
} else {
|
||||
// Emit forward direction memcpy then backward direction memcpy.
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : {1, -1}) {
|
||||
EmitMemcpy(Direction);
|
||||
if (Direction == 1) {
|
||||
@@ -2400,14 +2321,13 @@ DEF_OP(CacheLineClear) {
|
||||
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
|
||||
if (CTX->HostFeatures.DCacheSize() >= 64U) {
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, MemReg);
|
||||
} else {
|
||||
auto CurrentWorkingReg = MemReg.X();
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheSize()); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, CurrentWorkingReg);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheSize());
|
||||
for (size_t i = 0; i < std::max(1U, CTX->HostFeatures.DCacheLineSize / 64U); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
|
||||
CurrentWorkingReg = TMP1;
|
||||
}
|
||||
}
|
||||
@@ -2429,14 +2349,13 @@ DEF_OP(CacheLineClean) {
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
// Clean dcache only
|
||||
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
|
||||
if (CTX->HostFeatures.DCacheSize() >= 64U) {
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, MemReg);
|
||||
} else {
|
||||
auto CurrentWorkingReg = MemReg.X();
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheSize()); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, CurrentWorkingReg);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheSize());
|
||||
for (size_t i = 0; i < std::max(1U, CTX->HostFeatures.DCacheLineSize / 64U); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
|
||||
CurrentWorkingReg = TMP1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -15,7 +15,6 @@ $end_info$
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -48,10 +47,10 @@ DEF_OP(GuestOpcode) {
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
case IR::FenceType::Load: dmb(ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::FenceType::LoadStore: dmb(ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::FenceType::Store: dmb(ARMEmitter::BarrierScope::ST); break;
|
||||
case IR::FenceType::Inst: isb(); break;
|
||||
case IR::Fence_Load.Val: dmb(ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::Fence_LoadStore.Val: dmb(ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::Fence_Store.Val: dmb(ARMEmitter::BarrierScope::ST); break;
|
||||
case IR::Fence_Inst.Val: isb(); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
}
|
||||
}
|
||||
@@ -73,24 +72,24 @@ DEF_OP(Break) {
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case Core::FAULT_SIGILL:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGILL));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL));
|
||||
br(TMP1);
|
||||
break;
|
||||
case Core::FAULT_SIGTRAP:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGTRAP));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
break;
|
||||
case Core::FAULT_SIGSEGV:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGSEGV));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV));
|
||||
br(TMP1);
|
||||
break;
|
||||
default:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.GuestSignal_SIGTRAP));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
@@ -108,10 +107,10 @@ DEF_OP(GetRoundingMode) {
|
||||
// zero. Just swapping 01 and 10. That's a bitfield reverse. Round mode is in
|
||||
// bottom two bits. After reversing as a 32-bit operation, it'll be in [31:30]
|
||||
// and ripe for reinsertion back at 0.
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::Nearest) == 0);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::NegInfinity) == 1);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::PosInfinity) == 2);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::TowardsZero) == 3);
|
||||
static_assert(IR::ROUND_MODE_NEAREST == 0);
|
||||
static_assert(IR::ROUND_MODE_NEGATIVE_INFINITY == 1);
|
||||
static_assert(IR::ROUND_MODE_POSITIVE_INFINITY == 2);
|
||||
static_assert(IR::ROUND_MODE_TOWARDS_ZERO == 3);
|
||||
|
||||
rbit(ARMEmitter::Size::i32Bit, TMP1, Dst);
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP1, 30, 2);
|
||||
@@ -168,7 +167,7 @@ DEF_OP(PushRoundingMode) {
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(3 << 22));
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(Op->RoundMode << 22));
|
||||
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, (Op->RoundMode == 2 ? 1 : 2) << 22);
|
||||
}
|
||||
|
||||
@@ -189,11 +188,11 @@ DEF_OP(Print) {
|
||||
|
||||
if (IsGPR(Op->Value)) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.PrintValue));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue));
|
||||
} else {
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value), false);
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value), true);
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.PrintVectorValue));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
|
||||
}
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
@@ -210,25 +209,6 @@ DEF_OP(Print) {
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(PrintMsg) {
|
||||
auto Op = IROp->C<IR::IROp_PrintMsg>();
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, reinterpret_cast<uintptr_t>(Op->Value));
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.PrintMsgValue));
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
if (CTX->HostFeatures.SupportsCPUIndexInTPIDRRO) {
|
||||
mrs(GetReg(Node), ARMEmitter::SystemRegister::TPIDRRO_EL0);
|
||||
@@ -246,10 +226,7 @@ DEF_OP(ProcessorID) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(TMP1, {
|
||||
.GPRSpillMask = SpillMask,
|
||||
.FPRs = false,
|
||||
});
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -263,7 +240,7 @@ DEF_OP(ProcessorID) {
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Load the getcpu syscall number
|
||||
#if defined(ARCHITECTURE_x86_64)
|
||||
#if defined(_M_X86_64)
|
||||
// Just to ensure the syscall number doesn't change if compiled for an x86_64 host.
|
||||
constexpr auto GetCPUSyscallNum = 0xa8;
|
||||
#else
|
||||
@@ -282,17 +259,11 @@ DEF_OP(ProcessorID) {
|
||||
// Load the values returned by the kernel
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::WReg::w0, ARMEmitter::WReg::w1, ARMEmitter::Reg::rsp);
|
||||
// Deallocate stack space
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r8,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
.GPRFillMask = SpillMask,
|
||||
.FPRs = false,
|
||||
});
|
||||
FillStaticRegs(false, SpillMask, ~0U, ARMEmitter::Reg::r8, ARMEmitter::Reg::r2);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
@@ -333,20 +304,20 @@ DEF_OP(MonoBackpatcherWrite) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, TMP4);
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
|
||||
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
ldr(ARMEmitter::XReg::x4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.MonoBackpatcherWrite));
|
||||
ldr(ARMEmitter::XReg::x4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.MonoBackpatcherWrite));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, uint8_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4);
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
strb(ARMEmitter::WReg::zr, TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
@@ -18,4 +18,18 @@ DEF_OP(RMWHandle) {
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(IROp->Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(Swap1) {
|
||||
auto Op = IROp->C<IR::IROp_Swap1>();
|
||||
auto A = GetReg(Op->A), B = GetReg(Op->B);
|
||||
LOGMAN_THROW_A_FMT(B == GetReg(Node), "Invariant");
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, A);
|
||||
mov(ARMEmitter::Size::i64Bit, A, B);
|
||||
mov(ARMEmitter::Size::i64Bit, B, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Swap2) {
|
||||
// Implemented above
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,128 +1,79 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
enum class RelocationTypes : uint32_t {
|
||||
enum class RelocationTypes : uint8_t {
|
||||
// 8 byte literal in memory for symbol
|
||||
// Aligned to struct RelocNamedSymbolLiteral
|
||||
RELOC_NAMED_SYMBOL_LITERAL,
|
||||
|
||||
// Fixed size named thunk move
|
||||
// 4 instruction constant generation
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocNamedThunkMove
|
||||
RELOC_NAMED_THUNK_MOVE,
|
||||
|
||||
// 8 byte literal (relative to binary base address)
|
||||
RELOC_GUEST_RIP_LITERAL,
|
||||
|
||||
// Fixed size guest RIP move
|
||||
// 4 instruction constant generation
|
||||
// Aligned to struct RelocGuestRIP
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocGuestRIPMove
|
||||
RELOC_GUEST_RIP_MOVE,
|
||||
|
||||
// The frontend flagged those regions as patchable by the disk cache
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_DATA_MOVE,
|
||||
|
||||
// Same as GuestRipLiteral but patchable
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_RIP_LITERAL,
|
||||
|
||||
// Like PATCHABLE_RIP_LITERAL but puts it in a register
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_RIP_MOVE,
|
||||
};
|
||||
|
||||
struct FEX_PACKED RelocationHeader final {
|
||||
// Offset to the relocated host code data
|
||||
uint64_t Offset {};
|
||||
|
||||
struct RelocationTypeHeader final {
|
||||
RelocationTypes Type;
|
||||
};
|
||||
|
||||
struct RelocNamedSymbolLiteral final {
|
||||
enum class NamedSymbol : uint32_t {
|
||||
enum class NamedSymbol : uint8_t {
|
||||
///< Thread specific relocations
|
||||
// JIT Literal pointers
|
||||
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
|
||||
};
|
||||
|
||||
RelocationHeader Header {};
|
||||
RelocationTypeHeader Header {};
|
||||
|
||||
NamedSymbol Symbol;
|
||||
|
||||
uint32_t Pad[8];
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
};
|
||||
|
||||
struct RelocNamedThunkMove final {
|
||||
RelocationHeader Header {};
|
||||
RelocationTypeHeader Header {};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
uint32_t RegisterIndex;
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
// The thunk SHA256 hash
|
||||
IR::SHA256Sum Symbol;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
};
|
||||
|
||||
struct RelocGuestRIP final {
|
||||
RelocationHeader Header {};
|
||||
struct RelocGuestRIPMove final {
|
||||
RelocationTypeHeader Header {};
|
||||
|
||||
// GPR index the constant is being moved to (for non-literal relocations)
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
char Pad[3];
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
|
||||
// The base RIP (to be moved by the register for non-literal relocations).
|
||||
// In a serialized code cache, this is relative to the binary base address.
|
||||
// The unrelocated RIP that is being moved
|
||||
uint64_t GuestRIP;
|
||||
|
||||
uint32_t pad2[6] {};
|
||||
};
|
||||
|
||||
struct RelocGuestPatchableData final {
|
||||
RelocationHeader Header {};
|
||||
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
uint8_t ValueSize;
|
||||
|
||||
char Pad[2];
|
||||
|
||||
uint64_t SiteAddress;
|
||||
|
||||
uint32_t pad2[6] {};
|
||||
};
|
||||
|
||||
union Relocation {
|
||||
// Clang 16 Can't default-initialize this union
|
||||
static Relocation Default() {
|
||||
#if __clang_major__ < 17
|
||||
Relocation Ret {.Header {}};
|
||||
memset(&Ret, 0, sizeof(Ret));
|
||||
return Ret;
|
||||
#else
|
||||
return {};
|
||||
#endif
|
||||
}
|
||||
|
||||
RelocationHeader Header {};
|
||||
RelocationTypeHeader Header {};
|
||||
|
||||
RelocNamedSymbolLiteral NamedSymbolLiteral;
|
||||
// This makes our union of relocations at least 48 bytes
|
||||
// It might be more efficient to not use a union
|
||||
RelocNamedThunkMove NamedThunkMove;
|
||||
|
||||
RelocGuestRIP GuestRIP;
|
||||
|
||||
RelocGuestPatchableData GuestPatchableData;
|
||||
RelocGuestRIPMove GuestRIPMove;
|
||||
};
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl&, RelocNamedSymbolLiteral::NamedSymbol);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -41,7 +41,6 @@ namespace FEXCore::CPU {
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit; \
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
@@ -50,10 +49,8 @@ namespace FEXCore::CPU {
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else if (Is128Bit) { \
|
||||
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} else { \
|
||||
ARMOp(Dst.D(), Vector1.D(), Vector2.D()); \
|
||||
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} \
|
||||
}
|
||||
|
||||
@@ -747,11 +744,11 @@ DEF_OP(VFToIScalarInsert) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
|
||||
switch (RoundMode) {
|
||||
case IR::RoundMode::Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::Host: frinti(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Negative_Infinity: frintm(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Positive_Infinity: frintp(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Towards_Zero: frintz(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Host: frinti(SubRegSize.Scalar, Dst, Src); break;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -977,7 +974,7 @@ DEF_OP(LoadNamedVectorConstant) {
|
||||
}
|
||||
// Load the pointer.
|
||||
auto GenerateMemOperand = [this](IR::OpSize OpSize, uint32_t NamedConstant, ARMEmitter::Register Base) {
|
||||
const auto ConstantOffset = ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, Pointers.NamedVectorConstants, NamedConstant);
|
||||
const auto ConstantOffset = offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.NamedVectorConstants[NamedConstant]);
|
||||
|
||||
if (ConstantOffset <= 255 || // Unscaled 9-bit signed
|
||||
((ConstantOffset & (IR::OpSizeToSize(OpSize) - 1)) == 0 &&
|
||||
@@ -985,13 +982,13 @@ DEF_OP(LoadNamedVectorConstant) {
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, ConstantOffset);
|
||||
}
|
||||
|
||||
ldr(TMP1, STATE_PTR_IDX(CpuStateFrame, Pointers.NamedVectorConstantPointers, NamedConstant));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[NamedConstant]));
|
||||
return ARMEmitter::ExtendedMemOperand(TMP1, ARMEmitter::IndexType::OFFSET, 0);
|
||||
};
|
||||
|
||||
if (OpSize == IR::OpSize::i256Bit) {
|
||||
// Handle SVE 32-byte variant upfront.
|
||||
ldr(TMP1, STATE_PTR_IDX(CpuStateFrame, Pointers.NamedVectorConstantPointers, Op->Constant));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[Op->Constant]));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), TMP1, 0);
|
||||
return;
|
||||
}
|
||||
@@ -1013,7 +1010,7 @@ DEF_OP(LoadNamedVectorIndexedConstant) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
// Load the pointer.
|
||||
ldr(TMP1, STATE_PTR_IDX(CpuStateFrame, Pointers.IndexedNamedVectorConstantPointers, Op->Constant));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.IndexedNamedVectorConstantPointers[Op->Constant]));
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: ldrb(Dst, TMP1, Op->Index); break;
|
||||
@@ -1036,28 +1033,24 @@ DEF_OP(VMov) {
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Source = GetVReg(Op->Source);
|
||||
const auto Sub64BitHandler = [&](ARMEmitter::SubRegSize InsertSize) {
|
||||
if (Dst != Source) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
ins(InsertSize, Dst, 0, Source, 0);
|
||||
} else {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
ins(InsertSize, VTMP1, 0, Source, 0);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
};
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
Sub64BitHandler(ARMEmitter::SubRegSize::i8Bit);
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
ins(ARMEmitter::SubRegSize::i8Bit, VTMP1, 0, Source, 0);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i16Bit: {
|
||||
Sub64BitHandler(ARMEmitter::SubRegSize::i16Bit);
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 0, Source, 0);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
Sub64BitHandler(ARMEmitter::SubRegSize::i32Bit);
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, VTMP1, 0, Source, 0);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
@@ -1099,21 +1092,16 @@ DEF_OP(VAddP) {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
|
||||
// SVE ADDP is a destructive operation, so we need a temporary if
|
||||
// the destination and the lower vector don't alias.
|
||||
auto LHS = Dst;
|
||||
if (Dst != VectorLower) {
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
LHS = VTMP1;
|
||||
}
|
||||
// SVE ADDP is a destructive operation, so we need a temporary
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
|
||||
// Unlike Adv. SIMD's version of ADDP, which acts like it concats the
|
||||
// upper vector onto the end of the lower vector and then performs
|
||||
// pairwise addition, the SVE version actually interleaves the
|
||||
// results of the pairwise addition (gross!), so we need to undo that.
|
||||
addp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
|
||||
addp(SubRegSize, VTMP1.Z(), Pred, VTMP1.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
uzp2(SubRegSize, VTMP2.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
|
||||
// Merge upper half with lower half.
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
|
||||
@@ -1126,36 +1114,6 @@ DEF_OP(VAddP) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VOrn) {
|
||||
const auto Op = IROp->C<IR::IROp_VOrn>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst == Vector1) {
|
||||
bsl2n(Dst.Z(), Dst.Z(), Vector2.Z(), Dst.Z());
|
||||
} else if (Dst == Vector2) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
not_(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Pred, Dst.Z());
|
||||
orr(Dst.Z(), Vector1.Z(), Dst.Z());
|
||||
} else {
|
||||
movprfx(Dst.Z(), Vector1.Z());
|
||||
bsl2n(Dst.Z(), Dst.Z(), Vector2.Z(), Vector1.Z());
|
||||
}
|
||||
} else if (Is128Bit) {
|
||||
orn(Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
orn(Dst.D(), Vector1.D(), Vector2.D());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFAddV) {
|
||||
const auto Op = IROp->C<IR::IROp_VFAddV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1172,7 +1130,8 @@ DEF_OP(VFAddV) {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
|
||||
} else if (HostSupportsSVE128) {
|
||||
}
|
||||
if (HostSupportsSVE128) {
|
||||
const auto Pred = PRED_TMP_16B.Merging();
|
||||
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
|
||||
} else {
|
||||
@@ -1199,16 +1158,20 @@ DEF_OP(VAddV) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
uaddv(SubRegSize.Vector, Dst.D(), Mask, Vector.Z());
|
||||
} else {
|
||||
const auto Mask = ARMEmitter::PReg::p0;
|
||||
uaddv(SubRegSize.Vector, VTMP1.D(), Mask, Vector.Z());
|
||||
mov_imm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), 0);
|
||||
ptrue(SubRegSize.Vector, Mask, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Mask.Merging(), VTMP1.Z());
|
||||
}
|
||||
// SVE doesn't have an equivalent ADDV instruction, so we make do
|
||||
// by performing two Adv. SIMD ADDV operations on the high and low
|
||||
// 128-bit lanes and then sum them up.
|
||||
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto CompactPred = ARMEmitter::PReg::p0;
|
||||
|
||||
// Select all our upper elements to run ADDV over them.
|
||||
not_(CompactPred, Mask, PRED_TMP_16B);
|
||||
compact(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), CompactPred, Vector.Z());
|
||||
|
||||
addv(SubRegSize.Vector, VTMP2.Q(), Vector.Q());
|
||||
addv(SubRegSize.Vector, VTMP1.Q(), VTMP1.Q());
|
||||
add(SubRegSize.Vector, Dst.Q(), VTMP1.Q(), VTMP2.Q());
|
||||
} else {
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
addp(SubRegSize.Scalar, Dst, Vector);
|
||||
@@ -1297,7 +1260,6 @@ DEF_OP(VFAddP) {
|
||||
const auto Op = IROp->C<IR::IROp_VFAddP>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto IsScalar = OpSize == IR::OpSize::i64Bit;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
@@ -1310,26 +1272,19 @@ DEF_OP(VFAddP) {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
|
||||
// SVE FADDP is a destructive operation, so we need a temporary if
|
||||
// the destination and the lower vector don't alias.
|
||||
auto LHS = Dst;
|
||||
if (Dst != VectorLower) {
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
LHS = VTMP1;
|
||||
}
|
||||
// SVE FADDP is a destructive operation, so we need a temporary
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
|
||||
// Unlike Adv. SIMD's version of FADDP, which acts like it concats the
|
||||
// upper vector onto the end of the lower vector and then performs
|
||||
// pairwise addition, the SVE version actually interleaves the
|
||||
// results of the pairwise addition (gross!), so we need to undo that.
|
||||
faddp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
|
||||
faddp(SubRegSize, VTMP1.Z(), Pred, VTMP1.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
uzp2(SubRegSize, VTMP2.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
|
||||
// Merge upper half with lower half.
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
|
||||
} else if (IsScalar) {
|
||||
faddp(SubRegSize, Dst.D(), VectorLower.D(), VectorUpper.D());
|
||||
} else {
|
||||
faddp(SubRegSize, Dst.Q(), VectorLower.Q(), VectorUpper.Q());
|
||||
}
|
||||
@@ -1453,8 +1408,8 @@ DEF_OP(VFMin) {
|
||||
bif(Dst.Q(), Vector2.Q(), VTMP1.Q());
|
||||
} else if (Dst == Vector2) {
|
||||
// Destination is already Vector2, Invert arguments and insert Vector1 on false.
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
bit(Dst.Q(), Vector1.Q(), VTMP1.Q());
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bif(Dst.Q(), Vector1.Q(), VTMP1.Q());
|
||||
} else {
|
||||
// Dst is not either source, need a move.
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
@@ -1485,8 +1440,7 @@ DEF_OP(VFMax) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
fcmgt(SubRegSize, ComparePred, Mask.Zeroing(), Vector1.Z(), Vector2.Z());
|
||||
not_(ComparePred, Mask.Zeroing(), ComparePred);
|
||||
fcmgt(SubRegSize, ComparePred, Mask.Zeroing(), Vector2.Z(), Vector1.Z());
|
||||
|
||||
if (Dst == Vector1) {
|
||||
// Trivial case where Vector1 is also the destination.
|
||||
@@ -1508,17 +1462,17 @@ DEF_OP(VFMax) {
|
||||
|
||||
if (Dst == Vector1) {
|
||||
// Destination is already Vector1, need to insert Vector2 on true.
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bif(Dst.Q(), Vector2.Q(), VTMP1.Q());
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
bit(Dst.Q(), Vector2.Q(), VTMP1.Q());
|
||||
} else if (Dst == Vector2) {
|
||||
// Destination is already Vector2, Invert arguments and insert Vector1 on true.
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bit(Dst.Q(), Vector1.Q(), VTMP1.Q());
|
||||
} else {
|
||||
// Dst is not either source, need a move.
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
fcmgt(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
bif(Dst.Q(), Vector2.Q(), VTMP1.Q());
|
||||
bit(Dst.Q(), Vector2.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1545,14 +1499,9 @@ DEF_OP(VFRecp) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (Dst != Vector) {
|
||||
fmov(SubRegSize.Vector, Dst.Z(), 1.0);
|
||||
fdiv(SubRegSize.Vector, Dst.Z(), Pred, Dst.Z(), Vector.Z());
|
||||
} else {
|
||||
fmov(SubRegSize.Vector, VTMP1.Z(), 1.0);
|
||||
fdiv(SubRegSize.Vector, VTMP1.Z(), Pred, VTMP1.Z(), Vector.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
fmov(SubRegSize.Vector, VTMP1.Z(), 1.0);
|
||||
fdiv(SubRegSize.Vector, VTMP1.Z(), Pred, VTMP1.Z(), Vector.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
if (ElementSize == IR::OpSize::i32Bit && HostSupportsRPRES) {
|
||||
@@ -1804,14 +1753,10 @@ DEF_OP(VUMin) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
cmhi(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(Dst.Q(), Vector2.Q(), Vector1.Q());
|
||||
} else {
|
||||
cmhi(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
cmhi(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(VTMP2.Q(), Vector1.Q());
|
||||
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
|
||||
mov(Dst.Q(), VTMP2.Q());
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
@@ -1857,14 +1802,10 @@ DEF_OP(VSMin) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
cmgt(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(Dst.Q(), Vector2.Q(), Vector1.Q());
|
||||
} else {
|
||||
cmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
cmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
mov(VTMP2.Q(), Vector1.Q());
|
||||
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
|
||||
mov(Dst.Q(), VTMP2.Q());
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
@@ -1910,14 +1851,10 @@ DEF_OP(VUMax) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
cmhi(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
cmhi(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
cmhi(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(VTMP2.Q(), Vector1.Q());
|
||||
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
|
||||
mov(Dst.Q(), VTMP2.Q());
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
@@ -1963,14 +1900,10 @@ DEF_OP(VSMax) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
cmgt(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
cmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
cmgt(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(VTMP2.Q(), Vector1.Q());
|
||||
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
|
||||
mov(Dst.Q(), VTMP2.Q());
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
@@ -2790,17 +2723,17 @@ DEF_OP(VUShrSWide) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
lsr_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
|
||||
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
} else if (HostSupportsSVE128) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
@@ -2856,17 +2789,17 @@ DEF_OP(VSShrSWide) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
asr_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
|
||||
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
} else if (HostSupportsSVE128) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
@@ -2922,17 +2855,17 @@ DEF_OP(VUShlSWide) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
if (Dst != Vector) {
|
||||
// NOTE: SVE LSR is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
lsl_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
|
||||
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
} else if (HostSupportsSVE128) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
@@ -3020,14 +2953,9 @@ DEF_OP(VInsElement) {
|
||||
auto Reg = GetVReg(Op->DestVector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// Broadcast our source value across a temporary, then combine
|
||||
// with the destination.
|
||||
//
|
||||
// We don't need to perform the dup if we're just merging a 128-bit vector into
|
||||
// into an equivalent position since we have a predicate set up already.
|
||||
if (!(ElementSize == IR::OpSize::i128Bit && SrcIdx == DestIdx)) {
|
||||
dup(SubRegSize, VTMP2.Z(), SrcVector.Z(), SrcIdx);
|
||||
}
|
||||
// Broadcast our source value across a temporary,
|
||||
// then combine with the destination.
|
||||
dup(SubRegSize, VTMP2.Z(), SrcVector.Z(), SrcIdx);
|
||||
|
||||
// We don't need to move the data unnecessarily if
|
||||
// DestVector just so happens to also be the IR op
|
||||
@@ -3040,12 +2968,10 @@ DEF_OP(VInsElement) {
|
||||
|
||||
if (ElementSize == IR::OpSize::i128Bit) {
|
||||
if (DestIdx == 0) {
|
||||
const auto Source = SrcIdx == 0 ? SrcVector : VTMP2;
|
||||
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), PRED_TMP_16B.Merging(), Source.Z());
|
||||
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), PRED_TMP_16B.Merging(), VTMP2.Z());
|
||||
} else {
|
||||
const auto Source = SrcIdx == 1 ? SrcVector : VTMP2;
|
||||
not_(Predicate, PRED_TMP_32B.Zeroing(), PRED_TMP_16B);
|
||||
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Predicate.Merging(), Source.Z());
|
||||
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Predicate.Merging(), VTMP2.Z());
|
||||
}
|
||||
} else {
|
||||
const auto UpperBound = 16 >> FEXCore::ilog2(IR::OpSizeToSize(ElementSize));
|
||||
@@ -3180,12 +3106,19 @@ DEF_OP(VUShrI) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
} else {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (BitShift == 0) {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Z(), Vector.Z());
|
||||
}
|
||||
} else {
|
||||
lsr(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
// SVE LSR is destructive, so lets set up the destination if
|
||||
// Vector doesn't already alias it.
|
||||
if (Dst != Vector) {
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), BitShift);
|
||||
}
|
||||
} else {
|
||||
if (BitShift == 0) {
|
||||
@@ -3199,6 +3132,48 @@ DEF_OP(VUShrI) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUShraI) {
|
||||
const auto Op = IROp->C<IR::IROp_VUShraI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst == DestVector) {
|
||||
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
} else {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Z(), DestVector.Z());
|
||||
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
} else {
|
||||
mov(VTMP1.Z(), DestVector.Z());
|
||||
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (Dst == DestVector) {
|
||||
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
|
||||
} else {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Q(), DestVector.Q());
|
||||
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
|
||||
} else {
|
||||
mov(VTMP1.Q(), DestVector.Q());
|
||||
usra(SubRegSize, VTMP1.Q(), Vector.Q(), BitShift);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSShrI) {
|
||||
const auto Op = IROp->C<IR::IROp_VSShrI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -3214,12 +3189,19 @@ DEF_OP(VSShrI) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (Shift == 0) {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Z(), Vector.Z());
|
||||
}
|
||||
} else {
|
||||
asr(SubRegSize, Dst.Z(), Vector.Z(), Shift);
|
||||
// SVE ASR is destructive, so lets set up the destination if
|
||||
// Vector doesn't already alias it.
|
||||
if (Dst != Vector) {
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), Shift);
|
||||
}
|
||||
} else {
|
||||
if (Shift == 0) {
|
||||
@@ -3249,12 +3231,19 @@ DEF_OP(VShlI) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
} else {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
if (BitShift == 0) {
|
||||
if (Dst != Vector) {
|
||||
mov(Dst.Z(), Vector.Z());
|
||||
}
|
||||
} else {
|
||||
lsl(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
// SVE LSL is destructive, so lets set up the destination if
|
||||
// Vector doesn't already alias it.
|
||||
if (Dst != Vector) {
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), BitShift);
|
||||
}
|
||||
} else {
|
||||
if (BitShift == 0) {
|
||||
@@ -3281,13 +3270,8 @@ DEF_OP(VUShrNI) {
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (BitShift == 0) {
|
||||
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
|
||||
uzp1(SubRegSize, Dst.Z(), Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
}
|
||||
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
} else {
|
||||
if (BitShift == 0) {
|
||||
xtn(SubRegSize, Dst.D(), Vector.D());
|
||||
@@ -3335,55 +3319,6 @@ DEF_OP(VUShrNI2) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VRSHRN) {
|
||||
const auto Op = IROp->C<IR::IROp_VRSHRN>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
rshrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
} else {
|
||||
rshrn(SubRegSize, Dst.D(), Vector.D(), BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VRSHRNPair) {
|
||||
const auto Op = IROp->C<IR::IROp_VRSHRNPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
rshrnb(SubRegSize, VTMP1.Z(), VectorLower.Z(), BitShift);
|
||||
rshrnb(SubRegSize, VTMP2.Z(), VectorUpper.Z(), BitShift);
|
||||
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
} else {
|
||||
if (Dst == VectorUpper) {
|
||||
// RSHRN writes the lower half and would destroy the upper input.
|
||||
mov(VTMP1.Q(), VectorUpper.Q());
|
||||
VectorUpper = VTMP1;
|
||||
}
|
||||
|
||||
rshrn(SubRegSize, Dst.D(), VectorLower.D(), BitShift);
|
||||
rshrn2(SubRegSize, Dst.Q(), VectorUpper.Q(), BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSXTL) {
|
||||
const auto Op = IROp->C<IR::IROp_VSXTL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -3589,13 +3524,9 @@ DEF_OP(VSQXTN2) {
|
||||
mov(Dst.Q(), VectorLower.Q());
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, Dst, 1, VTMP2, 0);
|
||||
} else {
|
||||
if (Dst == VectorLower) {
|
||||
sqxtn2(SubRegSize, VectorLower, VectorUpper);
|
||||
} else {
|
||||
mov(VTMP1.Q(), VectorLower.Q());
|
||||
sqxtn2(SubRegSize, VTMP1, VectorUpper);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
mov(VTMP1.Q(), VectorLower.Q());
|
||||
sqxtn2(SubRegSize, VTMP1, VectorUpper);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -4451,9 +4382,13 @@ DEF_OP(VFMLS) {
|
||||
|
||||
if (Is128Bit) {
|
||||
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
|
||||
fmla(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
|
||||
}
|
||||
|
||||
if (Is128Bit) {
|
||||
fmla(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
fmla(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
|
||||
}
|
||||
|
||||
@@ -4473,7 +4408,7 @@ DEF_OP(VFNMLA) {
|
||||
// - SVE - FMLS
|
||||
// - ASIMD - FMLS
|
||||
// - Scalar - FMSUB
|
||||
const auto Op = IROp->C<IR::IROp_VFNMLA>();
|
||||
const auto Op = IROp->C<IR::IROp_VFMLA>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
@@ -4541,7 +4476,7 @@ DEF_OP(VFNMLS) {
|
||||
// - ASIMD - FMLS (With Negated addend)
|
||||
// - Scalar - FNMADD
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VFNMLS>();
|
||||
const auto Op = IROp->C<IR::IROp_VFMLS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
@@ -4607,9 +4542,13 @@ DEF_OP(VFNMLS) {
|
||||
|
||||
if (Is128Bit) {
|
||||
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
|
||||
fmls(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
|
||||
}
|
||||
|
||||
if (Is128Bit) {
|
||||
fmls(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
fmls(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
|
||||
}
|
||||
|
||||
@@ -4623,106 +4562,6 @@ DEF_OP(VFNMLS) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VBlendImm) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "Host must support SVE to use {}", __func__);
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VBlendImm>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto Selector = Op->Selector;
|
||||
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto LHS = GetVReg(Op->LHS);
|
||||
const auto RHS = GetVReg(Op->RHS);
|
||||
const auto DstIsNonAliasing = Dst != LHS && Dst != RHS;
|
||||
|
||||
// Silly case where two blending sources are the same.
|
||||
if (LHS == RHS) {
|
||||
if (DstIsNonAliasing) {
|
||||
mov(SubRegSize, Dst.Z(), GoverningPredicate.Merging(), LHS.Z());
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// We'll need to expand our selector to match its predicate equivalent.
|
||||
// The lowest bit of each predicate element being set to 1 signifies
|
||||
// that it's enabled.
|
||||
const auto MakePredicateMask = [ElementSize, Is256Bit, OpSize](uint16_t Imm) {
|
||||
if (ElementSize == IR::OpSize::i8Bit) {
|
||||
// Since we use a u16 selector, we have enough bits for every byte in a
|
||||
// 128-bit lane, so we don't need to do anything here except replicate the
|
||||
// bits in the event of 256-bit.
|
||||
return Is256Bit ? uint32_t(Imm) << 16 | Imm : Imm;
|
||||
}
|
||||
|
||||
uint32_t Mask = 0;
|
||||
const auto DataSize = IR::OpSizeToSize(ElementSize);
|
||||
const auto NumElements = IR::NumElements(OpSize, ElementSize);
|
||||
for (uint32_t i = 0; i < NumElements; i++) {
|
||||
if (((Imm >> i) & 1) != 0) {
|
||||
Mask |= 1U << (DataSize * i);
|
||||
}
|
||||
}
|
||||
return Mask;
|
||||
};
|
||||
|
||||
// Our predicate that we'll be firing our constructed bitmask into.
|
||||
constexpr auto Predicate = ARMEmitter::PReg::p0.Merging();
|
||||
|
||||
// TODO: We can completely eliminate this via PMOV in SVE2.1
|
||||
ARMEmitter::ForwardLabel AfterLabel;
|
||||
ARMEmitter::BackwardLabel ConstantLabel;
|
||||
(void)b(&AfterLabel);
|
||||
(void)Bind(&ConstantLabel);
|
||||
const auto PredicateMask = MakePredicateMask(Selector);
|
||||
if (Dst == RHS) {
|
||||
dc32(~PredicateMask);
|
||||
} else {
|
||||
dc32(PredicateMask);
|
||||
}
|
||||
(void)Bind(&AfterLabel);
|
||||
(void)adr(TMP1, &ConstantLabel);
|
||||
ldr(Predicate, TMP1);
|
||||
|
||||
if (Dst == LHS) {
|
||||
mov(SubRegSize, LHS.Z(), Predicate, RHS.Z());
|
||||
} else if (Dst == RHS) {
|
||||
mov(SubRegSize, RHS.Z(), Predicate, LHS.Z());
|
||||
} else {
|
||||
mov(SubRegSize, Dst.Z(), GoverningPredicate.Merging(), LHS.Z());
|
||||
mov(SubRegSize, Dst.Z(), Predicate, RHS.Z());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VXar) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "Host must support SVE to use {}", __func__);
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VXar>();
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(IROp->ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto LHS = GetVReg(Op->LHS);
|
||||
const auto RHS = GetVReg(Op->RHS);
|
||||
const auto Rotate = Op->Rotate;
|
||||
LOGMAN_THROW_A_FMT(Rotate >= 1 && Rotate <= ElementSizeBits, "Rotate immediate must be within [1, {}]", ElementSizeBits);
|
||||
|
||||
if (Dst == LHS) {
|
||||
xar(SubRegSize, Dst.Z(), RHS.Z(), Rotate);
|
||||
} else if (Dst == RHS) {
|
||||
movprfx(VTMP1.Z(), LHS.Z());
|
||||
xar(SubRegSize, VTMP1.Z(), RHS.Z(), Rotate);
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
} else {
|
||||
movprfx(Dst.Z(), LHS.Z());
|
||||
xar(SubRegSize, Dst.Z(), RHS.Z(), Rotate);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFCopySign) {
|
||||
auto Op = IROp->C<IR::IROp_VFCopySign>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -4746,150 +4585,4 @@ DEF_OP(VFCopySign) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM) {
|
||||
const auto Op = IROp->C<IR::IROp_F64FPREM>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FPREMHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM1) {
|
||||
const auto Op = IROp->C<IR::IROp_F64FPREM1>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FPREM1Handler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64SIN) {
|
||||
const auto Op = IROp->C<IR::IROp_F64SIN>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64SinHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64COS) {
|
||||
const auto Op = IROp->C<IR::IROp_F64COS>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64CosHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64TAN) {
|
||||
const auto Op = IROp->C<IR::IROp_F64TAN>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64TanHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
// Src1=y(ST1), Src2=x(ST0). Marshal into VTMP1/VTMP2 and dispatch the shared handler.
|
||||
DEF_OP(F64ATAN) {
|
||||
const auto Op = IROp->C<IR::IROp_F64ATAN>();
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64AtanHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
// Src=x(ST0), Src2=y(ST1). Marshal into VTMP1/VTMP2 and dispatch the shared handler.
|
||||
DEF_OP(F64FYL2X) {
|
||||
const auto Op = IROp->C<IR::IROp_F64FYL2X>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FYL2XHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
// Src=x(ST0), Src2=y(ST1). Marshal into VTMP1/VTMP2 and dispatch the shared handler.
|
||||
DEF_OP(F64FYL2XP1) {
|
||||
const auto Op = IROp->C<IR::IROp_F64FYL2XP1>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FYL2XP1Handler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64SCALE) {
|
||||
const auto Op = IROp->C<IR::IROp_F64SCALE>();
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64ScaleHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64F2XM1) {
|
||||
const auto Op = IROp->C<IR::IROp_F64F2XM1>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64F2XM1Handler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -15,7 +15,7 @@ $end_info$
|
||||
|
||||
namespace FEXCore {
|
||||
GuestToHostMap::GuestToHostMap()
|
||||
: BlockLinks_mbr {"FEXMem_BlockLinks"} {
|
||||
: BlockLinks_mbr {fextl::pmr::get_default_resource()} {
|
||||
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
|
||||
// Setup our PMR map.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
@@ -24,7 +24,7 @@ GuestToHostMap::GuestToHostMap()
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
: ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE + MAX_L1_SIZE;
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
|
||||
// Block cache ends up looking like this
|
||||
// PageMemoryMap[VirtualMemoryRegion >> 12]
|
||||
@@ -39,13 +39,6 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// We need one pointer per page of virtual memory
|
||||
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize, false, false));
|
||||
LOGMAN_THROW_A_FMT(PagePointer != -1ULL, "Failed to allocate PagePointer");
|
||||
|
||||
// Disable THP on the Lookup cache.
|
||||
FEXCore::Allocator::VirtualTHPControl(reinterpret_cast<const void*>(PagePointer), TotalCacheSize, FEXCore::Allocator::THPControl::Disable);
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Lookup", reinterpret_cast<void*>(PagePointer),
|
||||
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE);
|
||||
CTX->SyscallHandler->MarkOvercommitRange(PagePointer, TotalCacheSize);
|
||||
|
||||
// Allocate our memory backing our pages
|
||||
@@ -53,21 +46,14 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// XXX: We can drop down to 16KB if we store 4byte offsets from the code base
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8;
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / 4096 * 8;
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = PageMemory + CODE_SIZE;
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Lookup_L1", reinterpret_cast<void*>(L1Pointer), MAX_L1_SIZE);
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
|
||||
if (DynamicL1Cache()) {
|
||||
// Start at minimum size when dynamic.
|
||||
L1PointerMask = MIN_L1_ENTRIES - 1;
|
||||
} else {
|
||||
// Start at maximum instead.
|
||||
L1PointerMask = MAX_L1_ENTRIES - 1;
|
||||
}
|
||||
}
|
||||
|
||||
LookupCache::~LookupCache() {
|
||||
@@ -78,30 +64,31 @@ LookupCache::~LookupCache() {
|
||||
// These will get freed when their memory allocators are deallocated.
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache(const FEXCore::LookupCacheBaseLockToken& lk) {
|
||||
void LookupCache::ClearL2Cache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer),
|
||||
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE, false);
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
void LookupCache::ClearThreadLocalCaches(const LookupCacheWriteLockToken&) {
|
||||
// TODO: Preserve code cache entries?
|
||||
void LookupCache::ClearThreadLocalCaches() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
|
||||
// TODO: Rename this member to avoid confusion with code caching
|
||||
CachedCodePages.clear();
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache(const LookupCacheWriteLockToken& lk) {
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
ClearThreadLocalCaches(lk);
|
||||
Shared->ClearCache(lk);
|
||||
}
|
||||
|
||||
void GuestToHostMap::ClearCache(const LookupCacheWriteLockToken&) {
|
||||
void GuestToHostMap::ClearCache(const LockToken&) {
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
|
||||
@@ -2,58 +2,30 @@
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/SHMStats.h>
|
||||
#include <FEXCore/Utils/WritePriorityMutex.h>
|
||||
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/robin_set.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <span>
|
||||
#include <functional>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
struct LookupCacheBaseLockToken {
|
||||
protected:
|
||||
// Protected constructor - only derived classes can construct
|
||||
LookupCacheBaseLockToken() = default;
|
||||
};
|
||||
|
||||
struct LookupCacheWriteLockToken : public LookupCacheBaseLockToken {
|
||||
private:
|
||||
// Only constructible by GuestToHostMap
|
||||
friend struct GuestToHostMap;
|
||||
LookupCacheWriteLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
|
||||
: Lock {Mutex} {}
|
||||
std::lock_guard<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
|
||||
};
|
||||
|
||||
struct LookupCacheReadLockToken : public LookupCacheBaseLockToken {
|
||||
private:
|
||||
// Only constructible by GuestToHostMap
|
||||
friend struct GuestToHostMap;
|
||||
LookupCacheReadLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
|
||||
: Lock {Mutex} {}
|
||||
std::shared_lock<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
|
||||
};
|
||||
|
||||
struct GuestToHostMap {
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex Lock {};
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
struct LockToken {
|
||||
std::lock_guard<std::recursive_mutex> Lock;
|
||||
};
|
||||
|
||||
[[nodiscard]]
|
||||
LookupCacheWriteLockToken AcquireWriteLock() {
|
||||
return LookupCacheWriteLockToken {Lock};
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
LookupCacheReadLockToken AcquireReadLock() {
|
||||
return LookupCacheReadLockToken {Lock};
|
||||
LockToken AcquireLock() {
|
||||
return LockToken {std::lock_guard {WriteLock}};
|
||||
}
|
||||
|
||||
struct BlockLinkTag {
|
||||
@@ -77,74 +49,53 @@ struct GuestToHostMap {
|
||||
// walking each block member and destructing objects.
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
fextl::pmr::named_monotonic_page_buffer_resource BlockLinks_mbr;
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType* BlockLinks;
|
||||
|
||||
struct BlockEntry {
|
||||
uint64_t HostCode;
|
||||
fextl::vector<uint64_t> CodePages;
|
||||
};
|
||||
|
||||
fextl::robin_map<uint64_t, BlockEntry> BlockList;
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
||||
|
||||
GuestToHostMap();
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
const BlockEntry& AddBlockMapping(uint64_t Address, std::span<const uint64_t> CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode, const LockToken&) {
|
||||
// This may replace an existing mapping
|
||||
// NOTE: Generally no previous entry should exist, however there is one exception:
|
||||
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
||||
// may already contain the block address. Since is comparatively rare, we'll just leak
|
||||
// one of the two blocks in this case.
|
||||
return BlockList
|
||||
.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, fextl::vector<uint64_t>(CodePages.begin(), CodePages.end())})
|
||||
.first->second;
|
||||
BlockList[Address] = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
const BlockEntry* FindBlock(uint64_t Address, const LookupCacheReadLockToken&) {
|
||||
std::optional<uintptr_t> FindBlock(uint64_t Address, const LockToken&) {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
if (HostCode == BlockList.end()) {
|
||||
return nullptr;
|
||||
return std::nullopt;
|
||||
}
|
||||
return &HostCode->second;
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
bool Erase(uint64_t Address, const LookupCacheWriteLockToken&) {
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LockToken&) {
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second(it->first.HostLink);
|
||||
it->second(Frame, it->first.HostLink);
|
||||
}
|
||||
|
||||
// Remove from BlockList
|
||||
return BlockList.erase(Address) != 0;
|
||||
}
|
||||
|
||||
void InvalidateRange(uint64_t Start, uint64_t Length) {
|
||||
auto lk = AcquireWriteLock();
|
||||
|
||||
auto lower = CodePages.lower_bound(Start >> 12);
|
||||
auto upper = CodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (const auto& Entry : it->second) {
|
||||
Erase(Entry, lk);
|
||||
}
|
||||
}
|
||||
CodePages.erase(lower, upper);
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken&) {
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LockToken&) {
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
bool AddBlockExecutableRange(const std::ranges::input_range auto& Addresses, uint64_t Start, uint64_t Length, const LookupCacheWriteLockToken&) {
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LockToken&) {
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
@@ -156,7 +107,7 @@ struct GuestToHostMap {
|
||||
return rv;
|
||||
}
|
||||
|
||||
void ClearCache(const LookupCacheWriteLockToken&);
|
||||
void ClearCache(const LockToken&);
|
||||
};
|
||||
|
||||
class LookupCache {
|
||||
@@ -171,205 +122,122 @@ public:
|
||||
|
||||
// Swaps out the underlying GuestToHostMap and clears all associated caches.
|
||||
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap, const LookupCacheWriteLockToken& lk) {
|
||||
ClearThreadLocalCaches(lk);
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap) {
|
||||
ClearThreadLocalCaches();
|
||||
Shared = &NewMap;
|
||||
}
|
||||
|
||||
uintptr_t FindBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) {
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
// Try L1, no lock needed
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
// L2 and L3 need to be locked
|
||||
uintptr_t HostPtr {};
|
||||
{
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheReadLockTime : nullptr);
|
||||
auto lk = Shared->AcquireReadLock();
|
||||
LockTime.reset();
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
if (!DisableL2Cache()) {
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == Address) {
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
HostPtr = L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!HostPtr) {
|
||||
// Try L3
|
||||
auto Entry = Shared->FindBlock(Address, lk);
|
||||
if (Entry) {
|
||||
CacheBlockMapping(Address, *Entry, false, lk);
|
||||
HostPtr = Entry->HostCode;
|
||||
}
|
||||
if (BlockPointers[PageOffset].GuestCode == Address) {
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
|
||||
if (HostPtr && DynamicL1Cache()) {
|
||||
UpdateDynamicL1Stats(Thread, Address, HostPtr);
|
||||
// Try L3
|
||||
auto HostCode = Shared->FindBlock(Address, lk);
|
||||
if (HostCode) {
|
||||
CacheBlockMapping(Address, HostCode.value());
|
||||
return HostCode.value();
|
||||
}
|
||||
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedCacheMissCount, 1);
|
||||
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddress, uint64_t HostCode) {
|
||||
// If host pointer was found in L2 or L3, then add it to the counter.
|
||||
// Keeping track not L1 misses, but specifically L2/L3 hits.
|
||||
++L2L3CacheHits;
|
||||
|
||||
const auto CurrentTime = std::chrono::system_clock::now();
|
||||
const auto Period = CurrentTime - LastPeriod;
|
||||
if (Period >= SamplePeriod) {
|
||||
// If larger than the sample period then check if we need to increase L1 cache size.
|
||||
const double AveragePerSecond = static_cast<double>(L2L3CacheHits) /
|
||||
static_cast<double>(std::chrono::duration_cast<std::chrono::milliseconds>(Period).count()) * 1000.0;
|
||||
|
||||
if (AveragePerSecond >= DynamicL1CacheIncreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries < MAX_L1_ENTRIES) {
|
||||
// Entries whose address has the new mask bit set would be unreachable by InvalidateCache
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(L1Pointer), CurrentL1Entries * sizeof(LookupCacheEntry), false);
|
||||
|
||||
CurrentL1Entries <<= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
|
||||
// If L1 was just shrunk, then we just removed our cached entry. Add it back.
|
||||
AddL1Entry(GuestAddress, HostCode);
|
||||
}
|
||||
} else if (AveragePerSecond < DynamicL1CacheDecreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries > MIN_L1_ENTRIES) {
|
||||
CurrentL1Entries >>= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Madvise the entries that we are dropping. Gives the memory back to the OS.
|
||||
LookupCacheEntry* FirstZeroL1Entry = &reinterpret_cast<LookupCacheEntry*>(L1Pointer)[CurrentL1Entries];
|
||||
size_t ZeroMemorySize = (MAX_L1_ENTRIES - CurrentL1Entries) * sizeof(LookupCacheEntry);
|
||||
FEXCore::Allocator::VirtualDontNeed(FirstZeroL1Entry, ZeroMemorySize, false);
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
}
|
||||
}
|
||||
|
||||
// Update Last period to start again.
|
||||
LastPeriod = CurrentTime;
|
||||
L2L3CacheHits = 0;
|
||||
}
|
||||
// Failed to find
|
||||
return 0;
|
||||
}
|
||||
|
||||
GuestToHostMap* Shared = nullptr;
|
||||
|
||||
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, auto& Addresses, uint64_t Start, uint64_t Length) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
LockTime.reset();
|
||||
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
return Shared->AddBlockExecutableRange(Addresses, Start, Length, lk);
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, std::span<const uint64_t> CodePages, void* HostCode) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
LockTime.reset();
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
const auto& Entry = Shared->AddBlockMapping(Address, CodePages, HostCode, lk);
|
||||
Shared->AddBlockMapping(Address, HostCode, lk);
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
CacheBlockMapping(Address, Entry, true, lk);
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
// Invalidates L1/L2 for a given guest block
|
||||
void InvalidateCache(uint64_t Address, const LookupCacheWriteLockToken& lk) {
|
||||
// NOTE: It's the caller's responsibility to call Erase() for all other
|
||||
// GuestToHostMaps that share the same LookupCache. Otherwise, the
|
||||
// L1/L2 caches will contain stale references to deallocated memory.
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
bool ErasedAny = Shared->Erase(Frame, Address, lk);
|
||||
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = 0;
|
||||
ErasedAny = true;
|
||||
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
|
||||
// This is a soft guarantee for cross thread invalidation, as atomics are not used
|
||||
// and it hasn't been thoroughly tested
|
||||
}
|
||||
|
||||
if (!DisableL2Cache()) {
|
||||
// Do full map
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
// Do full map
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// Page for this code didn't even exist, nothing to do
|
||||
return;
|
||||
}
|
||||
|
||||
// Page exists, just set the offset to zero
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
BlockPointers[PageOffset].GuestCode = 0;
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// Page for this code didn't even exist, nothing to do
|
||||
return ErasedAny;
|
||||
}
|
||||
|
||||
// Page exists, just set the offset to zero
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
BlockPointers[PageOffset].GuestCode = 0;
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Invalidates all L1/L2 entries for all guest block that intersect the given range
|
||||
bool InvalidateCacheRange(uint64_t Start, uint64_t Length) {
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
|
||||
auto lower = CachedCodePages.lower_bound(Start >> 12);
|
||||
auto upper = CachedCodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (const auto& Entry : it->second) {
|
||||
InvalidateCache(Entry, lk);
|
||||
}
|
||||
}
|
||||
bool ret = upper != lower;
|
||||
CachedCodePages.erase(lower, upper);
|
||||
return ret;
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken& lk) {
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
|
||||
}
|
||||
|
||||
void ClearCache(const LookupCacheWriteLockToken&);
|
||||
void ClearL2Cache(const LookupCacheBaseLockToken&);
|
||||
void ClearThreadLocalCaches(const LookupCacheWriteLockToken&);
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
void ClearThreadLocalCaches();
|
||||
|
||||
uintptr_t GetL1Pointer() const {
|
||||
return L1Pointer;
|
||||
}
|
||||
uintptr_t GetScaledL1PointerMask() const {
|
||||
return L1PointerMask << FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry));
|
||||
}
|
||||
uintptr_t GetPagePointer() const {
|
||||
return PagePointer;
|
||||
}
|
||||
@@ -377,6 +245,9 @@ public:
|
||||
return VirtualMemSize;
|
||||
}
|
||||
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
@@ -384,56 +255,45 @@ public:
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
auto AcquireWriteLock() {
|
||||
return Shared->AcquireWriteLock();
|
||||
auto AcquireLock() {
|
||||
return Shared->AcquireLock();
|
||||
}
|
||||
|
||||
private:
|
||||
void AddL1Entry(uint64_t GuestAddress, uint64_t HostCode) {
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[GuestAddress & L1PointerMask];
|
||||
L1Entry.GuestCode = GuestAddress;
|
||||
L1Entry.HostCode = HostCode;
|
||||
}
|
||||
|
||||
void CacheBlockMapping(uint64_t Address, const GuestToHostMap::BlockEntry& Entry, bool L1Only, const LookupCacheBaseLockToken& lk) {
|
||||
for (const auto& CodePage : Entry.CodePages) {
|
||||
CachedCodePages[CodePage >> 12].insert(Address);
|
||||
}
|
||||
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
// Do L1
|
||||
AddL1Entry(Address, Entry.HostCode);
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = HostCode;
|
||||
|
||||
if (!DisableL2Cache() && !L1Only) {
|
||||
// Do ful map
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
// Do ful map
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
// Allocate one now if we can
|
||||
uintptr_t NewPageBacking = AllocateBackingForPage();
|
||||
if (!NewPageBacking) {
|
||||
// Couldn't allocate, clear L2 and retry
|
||||
ClearL2Cache(lk);
|
||||
CacheBlockMapping(FullAddress, Entry, false, lk);
|
||||
return;
|
||||
}
|
||||
Pointers[Address] = NewPageBacking;
|
||||
LocalPagePointer = NewPageBacking;
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
// Allocate one now if we can
|
||||
uintptr_t NewPageBacking = AllocateBackingForPage();
|
||||
if (!NewPageBacking) {
|
||||
// Couldn't allocate, clear L2 and retry
|
||||
ClearL2Cache();
|
||||
CacheBlockMapping(Address, HostCode);
|
||||
return;
|
||||
}
|
||||
|
||||
// Add the new pointer to the page block
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
// This silently replaces existing mappings
|
||||
BlockPointers[PageOffset].GuestCode = FullAddress;
|
||||
BlockPointers[PageOffset].HostCode = Entry.HostCode;
|
||||
Pointers[Address] = NewPageBacking;
|
||||
LocalPagePointer = NewPageBacking;
|
||||
}
|
||||
|
||||
// Add the new pointer to the page block
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
// This silently replaces existing mappings
|
||||
BlockPointers[PageOffset].GuestCode = FullAddress;
|
||||
BlockPointers[PageOffset].HostCode = HostCode;
|
||||
}
|
||||
|
||||
uintptr_t AllocateBackingForPage() {
|
||||
@@ -450,38 +310,19 @@ private:
|
||||
return PageMemory + NewBase;
|
||||
}
|
||||
|
||||
// Maps from a page index to all blocks in the page that have at some point been fetched into L1/L2
|
||||
fextl::map<uint64_t, fextl::robin_set<uint64_t>> CachedCodePages;
|
||||
|
||||
uintptr_t PagePointer;
|
||||
uintptr_t PageMemory;
|
||||
uintptr_t L1Pointer;
|
||||
uintptr_t L1PointerMask;
|
||||
|
||||
size_t TotalCacheSize;
|
||||
|
||||
// Start with 8k entries in L1 to give 128KB of L1 cache to each thread.
|
||||
// Max out at 1 million entries to give each thread 16MB of L1 cache maximum.
|
||||
constexpr static size_t MIN_L1_ENTRIES = 8 * 1024; // Must be a power of 2
|
||||
constexpr static size_t MAX_L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static size_t SIZE_PER_PAGE = FEXCore::Utils::FEX_PAGE_SIZE * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t MAX_L1_SIZE = MAX_L1_ENTRIES * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t L1_SIZE = L1_ENTRIES * sizeof(LookupCacheEntry);
|
||||
|
||||
size_t AllocateOffset {};
|
||||
|
||||
FEXCore::Context::ContextImpl* ctx;
|
||||
uint64_t VirtualMemSize {};
|
||||
|
||||
size_t CurrentL1Entries = MIN_L1_ENTRIES;
|
||||
uint64_t L2L3CacheHits {};
|
||||
std::chrono::time_point<std::chrono::system_clock> LastPeriod {};
|
||||
constexpr static std::chrono::seconds SamplePeriod {1};
|
||||
FEX_CONFIG_OPT(DynamicL1CacheIncreaseCountHeuristic, DYNAMICL1CACHEINCREASECOUNTHEURISTIC);
|
||||
FEX_CONFIG_OPT(DynamicL1CacheDecreaseCountHeuristic, DYNAMICL1CACHEDECREASECOUNTHEURISTIC);
|
||||
|
||||
FEX_CONFIG_OPT(DynamicL1Cache, DYNAMICL1CACHE);
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
} // namespace FEXCore
|
||||
File diff suppressed because it is too large.
Load diff
Loaded 100 of 1397 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user